{"timestamp_utc": "2026-04-12T22:14:34Z", "mode": "train", "global_step": 1, "epoch": 0.00010045203415369161, "loss": 0.1289, "grad_norm": 24.25541877746582, "learning_rate": 1e-05, "num_tokens": 1670.0, "completions/mean_length": 41.75, "completions/min_length": 29.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.75, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.6725290417671204, "rewards/meter/std": 0.40340036153793335, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9957022070884705, "rewards/repeat_soft/std": 0.00409209867939353, "rewards/judge_quality/mean": 0.4762499928474426, "rewards/judge_quality/std": 0.09941796213388443, "rewards/total_composite/mean": 0.6763333082199097, "rewards/total_composite/std": 0.20289066433906555, "reward": 0.6763333082199097, "reward_std": 0.20289064943790436, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22603146731853485, "sampling/sampling_logp_difference/max": 1.5626206398010254, "sampling/importance_sampling_ratio/min": 0.20958609879016876, "sampling/importance_sampling_ratio/mean": 1.013262391090393, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.01923768222332, "clip_ratio/low_mean": 0.08834622986614704, "clip_ratio/low_min": 0.08834622986614704, "clip_ratio/high_mean": 0.15415428206324577, "clip_ratio/high_max": 0.15415428206324577, "clip_ratio/region_mean": 0.24250051192939281, "reward_total_mean": 0.6763333082199097, "reward_meter_mean": 0.6725290417671204, "reward_meter_std": 0.40340036153793335, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9957022070884705, "reward_repeat_soft_std": 0.00409209867939353, "reward_judge_quality_mean": 0.4762499928474426, "reward_judge_quality_std": 0.09941796213388443, "reward_total_composite_mean": 0.6763333082199097, "reward_total_composite_std": 0.20289066433906555} {"timestamp_utc": "2026-04-12T22:14:41Z", "mode": "train", "global_step": 2, "epoch": 0.00020090406830738323, "loss": 0.0733, "grad_norm": 9.701464653015137, "learning_rate": 9.996969696969698e-06, "num_tokens": 4111.0, "completions/mean_length": 141.125, "completions/min_length": 106.0, "completions/max_length": 161.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 141.125, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 161.0, "rewards/meter/mean": 0.667778730392456, "rewards/meter/std": 0.3386266231536865, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9944028854370117, "rewards/repeat_soft/std": 0.00460322480648756, "rewards/judge_quality/mean": 0.6575000286102295, "rewards/judge_quality/std": 0.2133910059928894, "rewards/total_composite/mean": 0.7471907138824463, "rewards/total_composite/std": 0.150190070271492, "reward": 0.7471907138824463, "reward_std": 0.1501900553703308, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2034754604101181, "sampling/sampling_logp_difference/max": 1.743311882019043, "sampling/importance_sampling_ratio/min": 0.19116829335689545, "sampling/importance_sampling_ratio/mean": 1.0195162296295166, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9704072922468185, "clip_ratio/low_mean": 0.05871324986219406, "clip_ratio/low_min": 0.05871324986219406, "clip_ratio/high_mean": 0.128573689609766, "clip_ratio/high_max": 0.128573689609766, "clip_ratio/region_mean": 0.18728693947196007, "reward_total_mean": 0.7471907138824463, "reward_meter_mean": 0.667778730392456, "reward_meter_std": 0.3386266231536865, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9944028854370117, "reward_repeat_soft_std": 0.00460322480648756, "reward_judge_quality_mean": 0.6575000286102295, "reward_judge_quality_std": 0.2133910059928894, "reward_total_composite_mean": 0.7471907138824463, "reward_total_composite_std": 0.150190070271492} {"timestamp_utc": "2026-04-12T22:14:47Z", "mode": "train", "global_step": 3, "epoch": 0.00030135610246107485, "loss": 0.0512, "grad_norm": 21.948942184448242, "learning_rate": 9.993939393939395e-06, "num_tokens": 5819.0, "completions/mean_length": 36.5, "completions/min_length": 24.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.5, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.34533801674842834, "rewards/meter/std": 0.3437061011791229, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9806680679321289, "rewards/repeat_soft/std": 0.02873978205025196, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.25587037205696106, "rewards/total_composite/mean": 0.5868439078330994, "rewards/total_composite/std": 0.20601670444011688, "reward": 0.5868439078330994, "reward_std": 0.20601673424243927, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.319444864988327, "sampling/sampling_logp_difference/max": 2.2889962196350098, "sampling/importance_sampling_ratio/min": 0.10136815905570984, "sampling/importance_sampling_ratio/mean": 0.9693852066993713, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3289805501699448, "clip_ratio/low_mean": 0.15689394064247608, "clip_ratio/low_min": 0.15689394064247608, "clip_ratio/high_mean": 0.07748956605792046, "clip_ratio/high_max": 0.07748956605792046, "clip_ratio/region_mean": 0.23438350670039654, "reward_total_mean": 0.5868439078330994, "reward_meter_mean": 0.34533801674842834, "reward_meter_std": 0.3437061011791229, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9806680679321289, "reward_repeat_soft_std": 0.02873978205025196, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.25587037205696106, "reward_total_composite_mean": 0.5868439078330994, "reward_total_composite_std": 0.20601670444011688} {"timestamp_utc": "2026-04-12T22:14:53Z", "mode": "train", "global_step": 4, "epoch": 0.00040180813661476645, "loss": 0.0029, "grad_norm": 17.112354278564453, "learning_rate": 9.990909090909093e-06, "num_tokens": 7420.0, "completions/mean_length": 43.125, "completions/min_length": 35.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.125, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.6681925058364868, "rewards/meter/std": 0.34428924322128296, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9990890026092529, "rewards/repeat_soft/std": 0.00257677398622036, "rewards/judge_quality/mean": 0.7150000333786011, "rewards/judge_quality/std": 0.2377273440361023, "rewards/total_composite/mean": 0.7650955319404602, "rewards/total_composite/std": 0.1345381885766983, "reward": 0.7650955319404602, "reward_std": 0.1345381736755371, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21286356449127197, "sampling/sampling_logp_difference/max": 1.6740360260009766, "sampling/importance_sampling_ratio/min": 0.18748880922794342, "sampling/importance_sampling_ratio/mean": 1.0222951173782349, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9584360122680664, "clip_ratio/low_mean": 0.08466819114983082, "clip_ratio/low_min": 0.08466819114983082, "clip_ratio/high_mean": 0.13993905391544104, "clip_ratio/high_max": 0.13993905391544104, "clip_ratio/region_mean": 0.22460724506527185, "reward_total_mean": 0.7650955319404602, "reward_meter_mean": 0.6681925058364868, "reward_meter_std": 0.34428924322128296, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9990890026092529, "reward_repeat_soft_std": 0.00257677398622036, "reward_judge_quality_mean": 0.7150000333786011, "reward_judge_quality_std": 0.2377273440361023, "reward_total_composite_mean": 0.7650955319404602, "reward_total_composite_std": 0.1345381885766983} {"timestamp_utc": "2026-04-12T22:15:00Z", "mode": "train", "global_step": 5, "epoch": 0.0005022601707684581, "loss": 0.0895, "grad_norm": 9.455750465393066, "learning_rate": 9.987878787878788e-06, "num_tokens": 9686.0, "completions/mean_length": 106.25, "completions/min_length": 71.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.25, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.5534956455230713, "rewards/meter/std": 0.35001081228256226, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9964775443077087, "rewards/repeat_soft/std": 0.0026533447671681643, "rewards/judge_quality/mean": 0.7099999785423279, "rewards/judge_quality/std": 0.24213339388370514, "rewards/total_composite/mean": 0.7070332765579224, "rewards/total_composite/std": 0.182778000831604, "reward": 0.7070332765579224, "reward_std": 0.1827780157327652, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19017860293388367, "sampling/sampling_logp_difference/max": 1.9287948608398438, "sampling/importance_sampling_ratio/min": 0.14532321691513062, "sampling/importance_sampling_ratio/mean": 1.021316409111023, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3343858867883682, "clip_ratio/low_mean": 0.0882453341037035, "clip_ratio/low_min": 0.0882453341037035, "clip_ratio/high_mean": 0.0963182132691145, "clip_ratio/high_max": 0.0963182132691145, "clip_ratio/region_mean": 0.184563547372818, "reward_total_mean": 0.7070332765579224, "reward_meter_mean": 0.5534956455230713, "reward_meter_std": 0.35001081228256226, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9964775443077087, "reward_repeat_soft_std": 0.0026533447671681643, "reward_judge_quality_mean": 0.7099999785423279, "reward_judge_quality_std": 0.24213339388370514, "reward_total_composite_mean": 0.7070332765579224, "reward_total_composite_std": 0.182778000831604} {"timestamp_utc": "2026-04-12T22:15:07Z", "mode": "train", "global_step": 6, "epoch": 0.0006027122049221497, "loss": 0.1401, "grad_norm": 9.252427101135254, "learning_rate": 9.984848484848485e-06, "num_tokens": 12284.0, "completions/mean_length": 125.75, "completions/min_length": 85.0, "completions/max_length": 158.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.75, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 158.0, "rewards/meter/mean": 0.7442054748535156, "rewards/meter/std": 0.36003756523132324, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9989632368087769, "rewards/repeat_soft/std": 0.0006836142274551094, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.7295387983322144, "rewards/total_composite/std": 0.1836954951286316, "reward": 0.7295387983322144, "reward_std": 0.1836954802274704, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20029257237911224, "sampling/sampling_logp_difference/max": 1.7579917907714844, "sampling/importance_sampling_ratio/min": 0.17239071428775787, "sampling/importance_sampling_ratio/mean": 1.0128437280654907, "sampling/importance_sampling_ratio/max": 1.9965848922729492, "entropy": 2.4172347486019135, "clip_ratio/low_mean": 0.07391603104770184, "clip_ratio/low_min": 0.07391603104770184, "clip_ratio/high_mean": 0.1392985973507166, "clip_ratio/high_max": 0.1392985973507166, "clip_ratio/region_mean": 0.21321462839841843, "reward_total_mean": 0.7295387983322144, "reward_meter_mean": 0.7442054748535156, "reward_meter_std": 0.36003756523132324, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9989632368087769, "reward_repeat_soft_std": 0.0006836142274551094, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.7295387983322144, "reward_total_composite_std": 0.1836954951286316} {"timestamp_utc": "2026-04-12T22:15:14Z", "mode": "train", "global_step": 7, "epoch": 0.0007031642390758413, "loss": -0.0576, "grad_norm": 13.074331283569336, "learning_rate": 9.981818181818183e-06, "num_tokens": 14661.0, "completions/mean_length": 113.125, "completions/min_length": 84.0, "completions/max_length": 159.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.125, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 159.0, "rewards/meter/mean": 0.7239573001861572, "rewards/meter/std": 0.26627305150032043, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9978160858154297, "rewards/repeat_soft/std": 0.001829512882977724, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.7315623760223389, "rewards/total_composite/std": 0.12214145064353943, "reward": 0.7315623760223389, "reward_std": 0.12214145064353943, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2251056730747223, "sampling/sampling_logp_difference/max": 1.424403190612793, "sampling/importance_sampling_ratio/min": 0.24065205454826355, "sampling/importance_sampling_ratio/mean": 1.03545081615448, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.566889613866806, "clip_ratio/low_mean": 0.07710661552846432, "clip_ratio/low_min": 0.07710661552846432, "clip_ratio/high_mean": 0.1532231941819191, "clip_ratio/high_max": 0.1532231941819191, "clip_ratio/region_mean": 0.23032980971038342, "reward_total_mean": 0.7315623760223389, "reward_meter_mean": 0.7239573001861572, "reward_meter_std": 0.26627305150032043, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9978160858154297, "reward_repeat_soft_std": 0.001829512882977724, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.7315623760223389, "reward_total_composite_std": 0.12214145064353943} {"timestamp_utc": "2026-04-12T22:15:20Z", "mode": "train", "global_step": 8, "epoch": 0.0008036162732295329, "loss": 0.0692, "grad_norm": 37.71578598022461, "learning_rate": 9.97878787878788e-06, "num_tokens": 16079.0, "completions/mean_length": 21.25, "completions/min_length": 16.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.25, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.8553460836410522, "rewards/meter/std": 0.32826167345046997, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9081118702888489, "rewards/repeat_soft/std": 0.10070746392011642, "rewards/judge_quality/mean": 0.65625, "rewards/judge_quality/std": 0.29621124267578125, "rewards/total_composite/mean": 0.8225919008255005, "rewards/total_composite/std": 0.1914392113685608, "reward": 0.8225919008255005, "reward_std": 0.1914391964673996, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20113249123096466, "sampling/sampling_logp_difference/max": 1.2836437225341797, "sampling/importance_sampling_ratio/min": 0.2770260274410248, "sampling/importance_sampling_ratio/mean": 1.0172845125198364, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6074406504631042, "clip_ratio/low_mean": 0.06774115934967995, "clip_ratio/low_min": 0.06774115934967995, "clip_ratio/high_mean": 0.07261904887855053, "clip_ratio/high_max": 0.07261904887855053, "clip_ratio/region_mean": 0.14036020822823048, "reward_total_mean": 0.8225919008255005, "reward_meter_mean": 0.8553460836410522, "reward_meter_std": 0.32826167345046997, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9081118702888489, "reward_repeat_soft_std": 0.10070746392011642, "reward_judge_quality_mean": 0.65625, "reward_judge_quality_std": 0.29621124267578125, "reward_total_composite_mean": 0.8225919008255005, "reward_total_composite_std": 0.1914392113685608} {"timestamp_utc": "2026-04-12T22:15:27Z", "mode": "train", "global_step": 9, "epoch": 0.0009040683073832245, "loss": 0.0029, "grad_norm": 9.873559951782227, "learning_rate": 9.975757575757577e-06, "num_tokens": 18924.0, "completions/mean_length": 149.625, "completions/min_length": 105.0, "completions/max_length": 190.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 149.625, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 190.0, "rewards/meter/mean": 0.8943005800247192, "rewards/meter/std": 0.12937459349632263, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9977504014968872, "rewards/repeat_soft/std": 0.0031607486307621002, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.713111937046051, "rewards/total_composite/std": 0.29207462072372437, "reward": 0.713111937046051, "reward_std": 0.29207465052604675, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21067966520786285, "sampling/sampling_logp_difference/max": 1.7020273208618164, "sampling/importance_sampling_ratio/min": 0.18231353163719177, "sampling/importance_sampling_ratio/mean": 1.0195698738098145, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.251196339726448, "clip_ratio/low_mean": 0.01953125, "clip_ratio/low_min": 0.01953125, "clip_ratio/high_mean": 0.2094412874430418, "clip_ratio/high_max": 0.2094412874430418, "clip_ratio/region_mean": 0.2289725374430418, "reward_total_mean": 0.713111937046051, "reward_meter_mean": 0.8943005800247192, "reward_meter_std": 0.12937459349632263, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9977504014968872, "reward_repeat_soft_std": 0.0031607486307621002, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.713111937046051, "reward_total_composite_std": 0.29207462072372437} {"timestamp_utc": "2026-04-12T22:15:34Z", "mode": "train", "global_step": 10, "epoch": 0.0010045203415369162, "loss": 0.0833, "grad_norm": 25.13707733154297, "learning_rate": 9.972727272727274e-06, "num_tokens": 20493.0, "completions/mean_length": 34.125, "completions/min_length": 25.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.125, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.6848095059394836, "rewards/meter/std": 0.4096945524215698, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9840165376663208, "rewards/repeat_soft/std": 0.029735958203673363, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.6806565523147583, "rewards/total_composite/std": 0.34453243017196655, "reward": 0.6806565523147583, "reward_std": 0.34453243017196655, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21705979108810425, "sampling/sampling_logp_difference/max": 1.3225932121276855, "sampling/importance_sampling_ratio/min": 0.2664434611797333, "sampling/importance_sampling_ratio/mean": 1.0199761390686035, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.476714864373207, "clip_ratio/low_mean": 0.0765726137906313, "clip_ratio/low_min": 0.0765726137906313, "clip_ratio/high_mean": 0.11742378678172827, "clip_ratio/high_max": 0.11742378678172827, "clip_ratio/region_mean": 0.19399640057235956, "reward_total_mean": 0.6806565523147583, "reward_meter_mean": 0.6848095059394836, "reward_meter_std": 0.4096945524215698, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9840165376663208, "reward_repeat_soft_std": 0.029735958203673363, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.6806565523147583, "reward_total_composite_std": 0.34453243017196655} {"timestamp_utc": "2026-04-12T22:15:39Z", "mode": "train", "global_step": 11, "epoch": 0.0011049723756906078, "loss": 0.1787, "grad_norm": 18.680042266845703, "learning_rate": 9.96969696969697e-06, "num_tokens": 22069.0, "completions/mean_length": 47.0, "completions/min_length": 28.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.6532547473907471, "rewards/meter/std": 0.3094608187675476, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9933717846870422, "rewards/repeat_soft/std": 0.007688070647418499, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.7180517911911011, "rewards/total_composite/std": 0.15922589600086212, "reward": 0.7180517911911011, "reward_std": 0.15922589600086212, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20565736293792725, "sampling/sampling_logp_difference/max": 0.9617314338684082, "sampling/importance_sampling_ratio/min": 0.3822305202484131, "sampling/importance_sampling_ratio/mean": 1.0575566291809082, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3583329617977142, "clip_ratio/low_mean": 0.09450654499232769, "clip_ratio/low_min": 0.09450654499232769, "clip_ratio/high_mean": 0.12003943882882595, "clip_ratio/high_max": 0.12003943882882595, "clip_ratio/region_mean": 0.21454598382115364, "reward_total_mean": 0.7180517911911011, "reward_meter_mean": 0.6532547473907471, "reward_meter_std": 0.3094608187675476, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9933717846870422, "reward_repeat_soft_std": 0.007688070647418499, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.7180517911911011, "reward_total_composite_std": 0.15922589600086212} {"timestamp_utc": "2026-04-12T22:15:45Z", "mode": "train", "global_step": 12, "epoch": 0.0012054244098442994, "loss": 0.0339, "grad_norm": 28.459325790405273, "learning_rate": 9.966666666666667e-06, "num_tokens": 23560.0, "completions/mean_length": 26.375, "completions/min_length": 16.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.375, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.6143099069595337, "rewards/meter/std": 0.47742435336112976, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9266142845153809, "rewards/repeat_soft/std": 0.07501126825809479, "rewards/judge_quality/mean": 0.39375001192092896, "rewards/judge_quality/std": 0.09941794723272324, "rewards/total_composite/mean": 0.637225866317749, "rewards/total_composite/std": 0.23131398856639862, "reward": 0.637225866317749, "reward_std": 0.23131398856639862, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2075686752796173, "sampling/sampling_logp_difference/max": 1.0368385314941406, "sampling/importance_sampling_ratio/min": 0.3545738756656647, "sampling/importance_sampling_ratio/mean": 1.0258712768554688, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.740403950214386, "clip_ratio/low_mean": 0.06502757407724857, "clip_ratio/low_min": 0.06502757407724857, "clip_ratio/high_mean": 0.10130646172910929, "clip_ratio/high_max": 0.10130646172910929, "clip_ratio/region_mean": 0.16633403580635786, "reward_total_mean": 0.637225866317749, "reward_meter_mean": 0.6143099069595337, "reward_meter_std": 0.47742435336112976, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9266142845153809, "reward_repeat_soft_std": 0.07501126825809479, "reward_judge_quality_mean": 0.39375001192092896, "reward_judge_quality_std": 0.09941794723272324, "reward_total_composite_mean": 0.637225866317749, "reward_total_composite_std": 0.23131398856639862} {"timestamp_utc": "2026-04-12T22:15:51Z", "mode": "train", "global_step": 13, "epoch": 0.001305876443997991, "loss": -0.0121, "grad_norm": 21.01743507385254, "learning_rate": 9.963636363636364e-06, "num_tokens": 25175.0, "completions/mean_length": 44.875, "completions/min_length": 30.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.875, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.7244622707366943, "rewards/meter/std": 0.4260440170764923, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9886103868484497, "rewards/repeat_soft/std": 0.017639046534895897, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.6614029407501221, "rewards/total_composite/std": 0.2924328148365021, "reward": 0.6614029407501221, "reward_std": 0.2924328148365021, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21056580543518066, "sampling/sampling_logp_difference/max": 1.903547763824463, "sampling/importance_sampling_ratio/min": 0.1796707957983017, "sampling/importance_sampling_ratio/mean": 1.0298173427581787, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.938667967915535, "clip_ratio/low_mean": 0.06033376231789589, "clip_ratio/low_min": 0.06033376231789589, "clip_ratio/high_mean": 0.13869474548846483, "clip_ratio/high_max": 0.13869474548846483, "clip_ratio/region_mean": 0.19902850780636072, "reward_total_mean": 0.6614029407501221, "reward_meter_mean": 0.7244622707366943, "reward_meter_std": 0.4260440170764923, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9886103868484497, "reward_repeat_soft_std": 0.017639046534895897, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.6614029407501221, "reward_total_composite_std": 0.2924328148365021} {"timestamp_utc": "2026-04-12T22:15:58Z", "mode": "train", "global_step": 14, "epoch": 0.0014063284781516826, "loss": 0.1867, "grad_norm": 9.992451667785645, "learning_rate": 9.960606060606062e-06, "num_tokens": 27778.0, "completions/mean_length": 131.375, "completions/min_length": 89.0, "completions/max_length": 192.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.375, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 192.0, "rewards/meter/mean": 0.8847721815109253, "rewards/meter/std": 0.2510175108909607, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9979761838912964, "rewards/repeat_soft/std": 0.0023726883810013533, "rewards/judge_quality/mean": 0.5737500190734863, "rewards/judge_quality/std": 0.21836651861667633, "rewards/total_composite/mean": 0.8200700283050537, "rewards/total_composite/std": 0.16812320053577423, "reward": 0.8200700283050537, "reward_std": 0.16812318563461304, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2101648896932602, "sampling/sampling_logp_difference/max": 1.9604809284210205, "sampling/importance_sampling_ratio/min": 0.14079070091247559, "sampling/importance_sampling_ratio/mean": 1.0326483249664307, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4273063838481903, "clip_ratio/low_mean": 0.05659673362970352, "clip_ratio/low_min": 0.05659673362970352, "clip_ratio/high_mean": 0.15316884219646454, "clip_ratio/high_max": 0.15316884219646454, "clip_ratio/region_mean": 0.20976557582616806, "reward_total_mean": 0.8200700283050537, "reward_meter_mean": 0.8847721815109253, "reward_meter_std": 0.2510175108909607, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9979761838912964, "reward_repeat_soft_std": 0.0023726883810013533, "reward_judge_quality_mean": 0.5737500190734863, "reward_judge_quality_std": 0.21836651861667633, "reward_total_composite_mean": 0.8200700283050537, "reward_total_composite_std": 0.16812320053577423} {"timestamp_utc": "2026-04-12T22:16:04Z", "mode": "train", "global_step": 15, "epoch": 0.0015067805123053742, "loss": 0.0293, "grad_norm": 16.65377426147461, "learning_rate": 9.957575757575757e-06, "num_tokens": 29486.0, "completions/mean_length": 48.5, "completions/min_length": 25.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.5, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.4189527928829193, "rewards/meter/std": 0.3443324863910675, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9987918138504028, "rewards/repeat_soft/std": 0.00322684645652771, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.5490054488182068, "rewards/total_composite/std": 0.2890552580356598, "reward": 0.5490054488182068, "reward_std": 0.2890552580356598, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19479593634605408, "sampling/sampling_logp_difference/max": 2.340357780456543, "sampling/importance_sampling_ratio/min": 0.09629317373037338, "sampling/importance_sampling_ratio/mean": 1.0266320705413818, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6842268258333206, "clip_ratio/low_mean": 0.10162054747343063, "clip_ratio/low_min": 0.10162054747343063, "clip_ratio/high_mean": 0.11235708184540272, "clip_ratio/high_max": 0.11235708184540272, "clip_ratio/region_mean": 0.21397762931883335, "reward_total_mean": 0.5490054488182068, "reward_meter_mean": 0.4189527928829193, "reward_meter_std": 0.3443324863910675, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9987918138504028, "reward_repeat_soft_std": 0.00322684645652771, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.5490054488182068, "reward_total_composite_std": 0.2890552580356598} {"timestamp_utc": "2026-04-12T22:16:10Z", "mode": "train", "global_step": 16, "epoch": 0.0016072325464590658, "loss": -0.0032, "grad_norm": 16.885639190673828, "learning_rate": 9.954545454545456e-06, "num_tokens": 31057.0, "completions/mean_length": 45.375, "completions/min_length": 27.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.375, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.43825337290763855, "rewards/meter/std": 0.4620993435382843, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9892818331718445, "rewards/repeat_soft/std": 0.01204143650829792, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.5848921537399292, "rewards/total_composite/std": 0.20833450555801392, "reward": 0.5848921537399292, "reward_std": 0.20833447575569153, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.27420946955680847, "sampling/sampling_logp_difference/max": 2.7227797508239746, "sampling/importance_sampling_ratio/min": 0.06569189578294754, "sampling/importance_sampling_ratio/mean": 1.0040487051010132, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9974622428417206, "clip_ratio/low_mean": 0.07798896264284849, "clip_ratio/low_min": 0.07798896264284849, "clip_ratio/high_mean": 0.11831947416067123, "clip_ratio/high_max": 0.11831947416067123, "clip_ratio/region_mean": 0.19630843680351973, "reward_total_mean": 0.5848921537399292, "reward_meter_mean": 0.43825337290763855, "reward_meter_std": 0.4620993435382843, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9892818331718445, "reward_repeat_soft_std": 0.01204143650829792, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.5848921537399292, "reward_total_composite_std": 0.20833450555801392} {"timestamp_utc": "2026-04-12T22:16:15Z", "mode": "train", "global_step": 17, "epoch": 0.0017076845806127574, "loss": 0.081, "grad_norm": 18.933856964111328, "learning_rate": 9.951515151515152e-06, "num_tokens": 32706.0, "completions/mean_length": 47.125, "completions/min_length": 31.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.125, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.5159446001052856, "rewards/meter/std": 0.4337084889411926, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9935091733932495, "rewards/repeat_soft/std": 0.0063369860872626305, "rewards/judge_quality/mean": 0.5475000143051147, "rewards/judge_quality/std": 0.14320316910743713, "rewards/total_composite/mean": 0.6457759737968445, "rewards/total_composite/std": 0.2059926986694336, "reward": 0.6457759737968445, "reward_std": 0.2059926986694336, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21604323387145996, "sampling/sampling_logp_difference/max": 1.6152591705322266, "sampling/importance_sampling_ratio/min": 0.19883912801742554, "sampling/importance_sampling_ratio/mean": 1.0173397064208984, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0722051858901978, "clip_ratio/low_mean": 0.11042729578912258, "clip_ratio/low_min": 0.11042729578912258, "clip_ratio/high_mean": 0.11535557918250561, "clip_ratio/high_max": 0.11535557918250561, "clip_ratio/region_mean": 0.2257828749716282, "reward_total_mean": 0.6457759737968445, "reward_meter_mean": 0.5159446001052856, "reward_meter_std": 0.4337084889411926, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9935091733932495, "reward_repeat_soft_std": 0.0063369860872626305, "reward_judge_quality_mean": 0.5475000143051147, "reward_judge_quality_std": 0.14320316910743713, "reward_total_composite_mean": 0.6457759737968445, "reward_total_composite_std": 0.2059926986694336} {"timestamp_utc": "2026-04-12T22:16:21Z", "mode": "train", "global_step": 18, "epoch": 0.001808136614766449, "loss": -0.0019, "grad_norm": 17.924867630004883, "learning_rate": 9.948484848484849e-06, "num_tokens": 34138.0, "completions/mean_length": 30.0, "completions/min_length": 19.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.0, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.583729088306427, "rewards/meter/std": 0.350315123796463, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4087499976158142, "rewards/judge_quality/std": 0.10507649928331375, "rewards/total_composite/mean": 0.631553053855896, "rewards/total_composite/std": 0.17265585064888, "reward": 0.631553053855896, "reward_std": 0.17265585064888, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20634080469608307, "sampling/sampling_logp_difference/max": 1.4612393379211426, "sampling/importance_sampling_ratio/min": 0.2319486439228058, "sampling/importance_sampling_ratio/mean": 1.0241836309432983, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7248151153326035, "clip_ratio/low_mean": 0.10481186397373676, "clip_ratio/low_min": 0.10481186397373676, "clip_ratio/high_mean": 0.12524540536105633, "clip_ratio/high_max": 0.12524540536105633, "clip_ratio/region_mean": 0.2300572693347931, "reward_total_mean": 0.631553053855896, "reward_meter_mean": 0.583729088306427, "reward_meter_std": 0.350315123796463, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4087499976158142, "reward_judge_quality_std": 0.10507649928331375, "reward_total_composite_mean": 0.631553053855896, "reward_total_composite_std": 0.17265585064888} {"timestamp_utc": "2026-04-12T22:16:28Z", "mode": "train", "global_step": 19, "epoch": 0.0019085886489201406, "loss": -0.0696, "grad_norm": 14.65036392211914, "learning_rate": 9.945454545454546e-06, "num_tokens": 35798.0, "completions/mean_length": 52.5, "completions/min_length": 32.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.5, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.8749610185623169, "rewards/meter/std": 0.2477712482213974, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9977357387542725, "rewards/repeat_soft/std": 0.002847629599273205, "rewards/judge_quality/mean": 0.5550000071525574, "rewards/judge_quality/std": 0.33393755555152893, "rewards/total_composite/mean": 0.8100060224533081, "rewards/total_composite/std": 0.1606670320034027, "reward": 0.8100060224533081, "reward_std": 0.1606670320034027, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20267684757709503, "sampling/sampling_logp_difference/max": 1.836294412612915, "sampling/importance_sampling_ratio/min": 0.15940703451633453, "sampling/importance_sampling_ratio/mean": 1.0041242837905884, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5146896243095398, "clip_ratio/low_mean": 0.041744403541088104, "clip_ratio/low_min": 0.041744403541088104, "clip_ratio/high_mean": 0.1479868022724986, "clip_ratio/high_max": 0.1479868022724986, "clip_ratio/region_mean": 0.1897312058135867, "reward_total_mean": 0.8100060224533081, "reward_meter_mean": 0.8749610185623169, "reward_meter_std": 0.2477712482213974, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9977357387542725, "reward_repeat_soft_std": 0.002847629599273205, "reward_judge_quality_mean": 0.5550000071525574, "reward_judge_quality_std": 0.33393755555152893, "reward_total_composite_mean": 0.8100060224533081, "reward_total_composite_std": 0.1606670320034027} {"timestamp_utc": "2026-04-12T22:16:35Z", "mode": "train", "global_step": 20, "epoch": 0.0020090406830738324, "loss": 0.1452, "grad_norm": 12.367934226989746, "learning_rate": 9.942424242424244e-06, "num_tokens": 38169.0, "completions/mean_length": 109.375, "completions/min_length": 66.0, "completions/max_length": 165.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.375, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 165.0, "rewards/meter/mean": 0.7363971471786499, "rewards/meter/std": 0.274208128452301, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.2519763112068176, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9973340034484863, "rewards/repeat_soft/std": 0.002601567655801773, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.6832370758056641, "rewards/total_composite/std": 0.1437705010175705, "reward": 0.6832370758056641, "reward_std": 0.1437705010175705, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20733380317687988, "sampling/sampling_logp_difference/max": 1.9372377395629883, "sampling/importance_sampling_ratio/min": 0.14410145580768585, "sampling/importance_sampling_ratio/mean": 1.012389898300171, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.014356479048729, "clip_ratio/low_mean": 0.09246364235877991, "clip_ratio/low_min": 0.09246364235877991, "clip_ratio/high_mean": 0.11669936217367649, "clip_ratio/high_max": 0.11669936217367649, "clip_ratio/region_mean": 0.2091630045324564, "reward_total_mean": 0.6832370758056641, "reward_meter_mean": 0.7363971471786499, "reward_meter_std": 0.274208128452301, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.2519763112068176, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9973340034484863, "reward_repeat_soft_std": 0.002601567655801773, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.6832370758056641, "reward_total_composite_std": 0.1437705010175705} {"timestamp_utc": "2026-04-12T22:16:40Z", "mode": "train", "global_step": 21, "epoch": 0.002109492717227524, "loss": 0.0954, "grad_norm": 33.46713638305664, "learning_rate": 9.939393939393939e-06, "num_tokens": 39729.0, "completions/mean_length": 36.0, "completions/min_length": 26.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.0, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.24060703814029694, "rewards/meter/std": 0.28769800066947937, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9997790455818176, "rewards/repeat_soft/std": 0.0006249745492823422, "rewards/judge_quality/mean": 0.7987500429153442, "rewards/judge_quality/std": 0.22465451061725616, "rewards/total_composite/mean": 0.5978760719299316, "rewards/total_composite/std": 0.15713509917259216, "reward": 0.5978760719299316, "reward_std": 0.15713508427143097, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2637569010257721, "sampling/sampling_logp_difference/max": 2.1834075450897217, "sampling/importance_sampling_ratio/min": 0.11265699565410614, "sampling/importance_sampling_ratio/mean": 1.0002814531326294, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2168454974889755, "clip_ratio/low_mean": 0.18306511640548706, "clip_ratio/low_min": 0.18306511640548706, "clip_ratio/high_mean": 0.03896103985607624, "clip_ratio/high_max": 0.03896103985607624, "clip_ratio/region_mean": 0.2220261562615633, "reward_total_mean": 0.5978760719299316, "reward_meter_mean": 0.24060703814029694, "reward_meter_std": 0.28769800066947937, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9997790455818176, "reward_repeat_soft_std": 0.0006249745492823422, "reward_judge_quality_mean": 0.7987500429153442, "reward_judge_quality_std": 0.22465451061725616, "reward_total_composite_mean": 0.5978760719299316, "reward_total_composite_std": 0.15713509917259216} {"timestamp_utc": "2026-04-12T22:16:46Z", "mode": "train", "global_step": 22, "epoch": 0.0022099447513812156, "loss": 0.0519, "grad_norm": 14.068692207336426, "learning_rate": 9.936363636363638e-06, "num_tokens": 41479.0, "completions/mean_length": 59.75, "completions/min_length": 39.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.75, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9757623672485352, "rewards/meter/std": 0.020018933340907097, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9967460632324219, "rewards/repeat_soft/std": 0.005034734960645437, "rewards/judge_quality/mean": 0.6024999618530273, "rewards/judge_quality/std": 0.2521762549877167, "rewards/total_composite/mean": 0.8695176839828491, "rewards/total_composite/std": 0.07778604328632355, "reward": 0.8695176839828491, "reward_std": 0.07778605818748474, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20508892834186554, "sampling/sampling_logp_difference/max": 1.620192527770996, "sampling/importance_sampling_ratio/min": 0.19786059856414795, "sampling/importance_sampling_ratio/mean": 1.0207918882369995, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.071342244744301, "clip_ratio/low_mean": 0.10649552661925554, "clip_ratio/low_min": 0.10649552661925554, "clip_ratio/high_mean": 0.07769765146076679, "clip_ratio/high_max": 0.07769765146076679, "clip_ratio/region_mean": 0.18419317808002234, "reward_total_mean": 0.8695176839828491, "reward_meter_mean": 0.9757623672485352, "reward_meter_std": 0.020018933340907097, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9967460632324219, "reward_repeat_soft_std": 0.005034734960645437, "reward_judge_quality_mean": 0.6024999618530273, "reward_judge_quality_std": 0.2521762549877167, "reward_total_composite_mean": 0.8695176839828491, "reward_total_composite_std": 0.07778604328632355} {"timestamp_utc": "2026-04-12T22:16:52Z", "mode": "train", "global_step": 23, "epoch": 0.0023103967855349072, "loss": 0.0951, "grad_norm": 25.26443099975586, "learning_rate": 9.933333333333334e-06, "num_tokens": 43076.0, "completions/mean_length": 39.625, "completions/min_length": 35.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.625, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.41748276352882385, "rewards/meter/std": 0.4426417350769043, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9921599626541138, "rewards/repeat_soft/std": 0.009428190067410469, "rewards/judge_quality/mean": 0.5612500309944153, "rewards/judge_quality/std": 0.1968638300895691, "rewards/total_composite/mean": 0.6054582595825195, "rewards/total_composite/std": 0.19344164431095123, "reward": 0.6054582595825195, "reward_std": 0.19344164431095123, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1857699751853943, "sampling/sampling_logp_difference/max": 1.0207924842834473, "sampling/importance_sampling_ratio/min": 0.36030930280685425, "sampling/importance_sampling_ratio/mean": 1.0526957511901855, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6756559312343597, "clip_ratio/low_mean": 0.11134098190814257, "clip_ratio/low_min": 0.11134098190814257, "clip_ratio/high_mean": 0.04910453222692013, "clip_ratio/high_max": 0.04910453222692013, "clip_ratio/region_mean": 0.1604455141350627, "reward_total_mean": 0.6054582595825195, "reward_meter_mean": 0.41748276352882385, "reward_meter_std": 0.4426417350769043, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9921599626541138, "reward_repeat_soft_std": 0.009428190067410469, "reward_judge_quality_mean": 0.5612500309944153, "reward_judge_quality_std": 0.1968638300895691, "reward_total_composite_mean": 0.6054582595825195, "reward_total_composite_std": 0.19344164431095123} {"timestamp_utc": "2026-04-12T22:16:58Z", "mode": "train", "global_step": 24, "epoch": 0.002410848819688599, "loss": 0.0564, "grad_norm": 18.268474578857422, "learning_rate": 9.930303030303031e-06, "num_tokens": 44791.0, "completions/mean_length": 62.375, "completions/min_length": 42.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.375, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.634819746017456, "rewards/meter/std": 0.4580250084400177, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9998894929885864, "rewards/repeat_soft/std": 0.00031247673905454576, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.7179077863693237, "rewards/total_composite/std": 0.2593117356300354, "reward": 0.7179077863693237, "reward_std": 0.2593117356300354, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21290218830108643, "sampling/sampling_logp_difference/max": 1.9016647338867188, "sampling/importance_sampling_ratio/min": 0.1493198424577713, "sampling/importance_sampling_ratio/mean": 1.0068492889404297, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.353201448917389, "clip_ratio/low_mean": 0.06295787636190653, "clip_ratio/low_min": 0.06295787636190653, "clip_ratio/high_mean": 0.107818647287786, "clip_ratio/high_max": 0.107818647287786, "clip_ratio/region_mean": 0.17077652364969254, "reward_total_mean": 0.7179077863693237, "reward_meter_mean": 0.634819746017456, "reward_meter_std": 0.4580250084400177, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9998894929885864, "reward_repeat_soft_std": 0.00031247673905454576, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.7179077863693237, "reward_total_composite_std": 0.2593117356300354} {"timestamp_utc": "2026-04-12T22:17:05Z", "mode": "train", "global_step": 25, "epoch": 0.0025113008538422904, "loss": 0.0743, "grad_norm": 8.258381843566895, "learning_rate": 9.927272727272728e-06, "num_tokens": 47647.0, "completions/mean_length": 148.0, "completions/min_length": 129.0, "completions/max_length": 173.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 148.0, "completions/min_terminated_length": 129.0, "completions/max_terminated_length": 173.0, "rewards/meter/mean": 0.5918176174163818, "rewards/meter/std": 0.33095410466194153, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9976328611373901, "rewards/repeat_soft/std": 0.0029239931609481573, "rewards/judge_quality/mean": 0.7024999856948853, "rewards/judge_quality/std": 0.24294327199459076, "rewards/total_composite/mean": 0.7193312048912048, "rewards/total_composite/std": 0.1843627542257309, "reward": 0.7193312048912048, "reward_std": 0.1843627244234085, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16058434545993805, "sampling/sampling_logp_difference/max": 2.274415969848633, "sampling/importance_sampling_ratio/min": 0.10285695642232895, "sampling/importance_sampling_ratio/mean": 1.0136536359786987, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0220863968133926, "clip_ratio/low_mean": 0.06519535928964615, "clip_ratio/low_min": 0.06519535928964615, "clip_ratio/high_mean": 0.06505231745541096, "clip_ratio/high_max": 0.06505231745541096, "clip_ratio/region_mean": 0.1302476767450571, "reward_total_mean": 0.7193312048912048, "reward_meter_mean": 0.5918176174163818, "reward_meter_std": 0.33095410466194153, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9976328611373901, "reward_repeat_soft_std": 0.0029239931609481573, "reward_judge_quality_mean": 0.7024999856948853, "reward_judge_quality_std": 0.24294327199459076, "reward_total_composite_mean": 0.7193312048912048, "reward_total_composite_std": 0.1843627542257309} {"timestamp_utc": "2026-04-12T22:17:12Z", "mode": "train", "global_step": 26, "epoch": 0.002611752887995982, "loss": -0.1294, "grad_norm": 24.321762084960938, "learning_rate": 9.924242424242425e-06, "num_tokens": 49143.0, "completions/mean_length": 30.0, "completions/min_length": 18.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.0, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9500245451927185, "rewards/meter/std": 0.13721725344657898, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7997610569000244, "rewards/total_composite/std": 0.06174775958061218, "reward": 0.7997610569000244, "reward_std": 0.06174774840474129, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0748242437839508, "sampling/sampling_logp_difference/max": 1.1188440322875977, "sampling/importance_sampling_ratio/min": 0.32665717601776123, "sampling/importance_sampling_ratio/mean": 0.9947167038917542, "sampling/importance_sampling_ratio/max": 1.8977136611938477, "entropy": 0.4834756925702095, "clip_ratio/low_mean": 0.02777777798473835, "clip_ratio/low_min": 0.02777777798473835, "clip_ratio/high_mean": 0.030977724120020866, "clip_ratio/high_max": 0.030977724120020866, "clip_ratio/region_mean": 0.058755502104759216, "reward_total_mean": 0.7997610569000244, "reward_meter_mean": 0.9500245451927185, "reward_meter_std": 0.13721725344657898, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7997610569000244, "reward_total_composite_std": 0.06174775958061218} {"timestamp_utc": "2026-04-12T22:17:19Z", "mode": "train", "global_step": 27, "epoch": 0.0027122049221496736, "loss": -0.0948, "grad_norm": 7.565678596496582, "learning_rate": 9.921212121212121e-06, "num_tokens": 51958.0, "completions/mean_length": 154.875, "completions/min_length": 95.0, "completions/max_length": 189.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 154.875, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 189.0, "rewards/meter/mean": 0.611936628818512, "rewards/meter/std": 0.3855309784412384, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9979890584945679, "rewards/repeat_soft/std": 0.0016030868282541633, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5971717834472656, "rewards/total_composite/std": 0.28371596336364746, "reward": 0.5971717834472656, "reward_std": 0.28371599316596985, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21463504433631897, "sampling/sampling_logp_difference/max": 2.0930118560791016, "sampling/importance_sampling_ratio/min": 0.12331517040729523, "sampling/importance_sampling_ratio/mean": 1.0371588468551636, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4359234273433685, "clip_ratio/low_mean": 0.07277752831578255, "clip_ratio/low_min": 0.07277752831578255, "clip_ratio/high_mean": 0.1326075941324234, "clip_ratio/high_max": 0.1326075941324234, "clip_ratio/region_mean": 0.20538512244820595, "reward_total_mean": 0.5971717834472656, "reward_meter_mean": 0.611936628818512, "reward_meter_std": 0.3855309784412384, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9979890584945679, "reward_repeat_soft_std": 0.0016030868282541633, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5971717834472656, "reward_total_composite_std": 0.28371596336364746} {"timestamp_utc": "2026-04-12T22:17:24Z", "mode": "train", "global_step": 28, "epoch": 0.0028126569563033652, "loss": 0.0946, "grad_norm": 23.491594314575195, "learning_rate": 9.918181818181818e-06, "num_tokens": 53496.0, "completions/mean_length": 30.25, "completions/min_length": 20.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.25, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.5963934063911438, "rewards/meter/std": 0.4609495997428894, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9303162097930908, "rewards/repeat_soft/std": 0.06846698373556137, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.644158661365509, "rewards/total_composite/std": 0.20341292023658752, "reward": 0.644158661365509, "reward_std": 0.20341289043426514, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22669285535812378, "sampling/sampling_logp_difference/max": 1.899033546447754, "sampling/importance_sampling_ratio/min": 0.14971324801445007, "sampling/importance_sampling_ratio/mean": 1.014909267425537, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6166850328445435, "clip_ratio/low_mean": 0.0859464481472969, "clip_ratio/low_min": 0.0859464481472969, "clip_ratio/high_mean": 0.08879269333556294, "clip_ratio/high_max": 0.08879269333556294, "clip_ratio/region_mean": 0.17473914148285985, "reward_total_mean": 0.644158661365509, "reward_meter_mean": 0.5963934063911438, "reward_meter_std": 0.4609495997428894, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9303162097930908, "reward_repeat_soft_std": 0.06846698373556137, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.644158661365509, "reward_total_composite_std": 0.20341292023658752} {"timestamp_utc": "2026-04-12T22:17:32Z", "mode": "train", "global_step": 29, "epoch": 0.002913108990457057, "loss": 0.0661, "grad_norm": 7.243879795074463, "learning_rate": 9.915151515151515e-06, "num_tokens": 56603.0, "completions/mean_length": 188.375, "completions/min_length": 179.0, "completions/max_length": 204.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 188.375, "completions/min_terminated_length": 179.0, "completions/max_terminated_length": 204.0, "rewards/meter/mean": 0.6960962414741516, "rewards/meter/std": 0.3066192865371704, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9989635348320007, "rewards/repeat_soft/std": 0.000937586824875325, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.6790146827697754, "rewards/total_composite/std": 0.14062994718551636, "reward": 0.6790146827697754, "reward_std": 0.14062994718551636, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19934551417827606, "sampling/sampling_logp_difference/max": 1.6785790920257568, "sampling/importance_sampling_ratio/min": 0.1866389811038971, "sampling/importance_sampling_ratio/mean": 1.0356954336166382, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.226356267929077, "clip_ratio/low_mean": 0.06664236634969711, "clip_ratio/low_min": 0.06664236634969711, "clip_ratio/high_mean": 0.12853810749948025, "clip_ratio/high_max": 0.12853810749948025, "clip_ratio/region_mean": 0.19518047384917736, "reward_total_mean": 0.6790146827697754, "reward_meter_mean": 0.6960962414741516, "reward_meter_std": 0.3066192865371704, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9989635348320007, "reward_repeat_soft_std": 0.000937586824875325, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.6790146827697754, "reward_total_composite_std": 0.14062994718551636} {"timestamp_utc": "2026-04-12T22:17:38Z", "mode": "train", "global_step": 30, "epoch": 0.0030135610246107484, "loss": 0.0748, "grad_norm": 11.622523307800293, "learning_rate": 9.912121212121213e-06, "num_tokens": 58613.0, "completions/mean_length": 67.25, "completions/min_length": 48.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.25, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.6762853264808655, "rewards/meter/std": 0.3140008747577667, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9974287152290344, "rewards/repeat_soft/std": 0.003936082124710083, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.6988212466239929, "rewards/total_composite/std": 0.10489797592163086, "reward": 0.6988212466239929, "reward_std": 0.10489796847105026, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20940160751342773, "sampling/sampling_logp_difference/max": 2.6506054401397705, "sampling/importance_sampling_ratio/min": 0.07060845196247101, "sampling/importance_sampling_ratio/mean": 1.0177046060562134, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0744355469942093, "clip_ratio/low_mean": 0.08076691627502441, "clip_ratio/low_min": 0.08076691627502441, "clip_ratio/high_mean": 0.1163441464304924, "clip_ratio/high_max": 0.1163441464304924, "clip_ratio/region_mean": 0.19711106270551682, "reward_total_mean": 0.6988212466239929, "reward_meter_mean": 0.6762853264808655, "reward_meter_std": 0.3140008747577667, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9974287152290344, "reward_repeat_soft_std": 0.003936082124710083, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.6988212466239929, "reward_total_composite_std": 0.10489797592163086} {"timestamp_utc": "2026-04-12T22:17:45Z", "mode": "train", "global_step": 31, "epoch": 0.00311401305876444, "loss": 0.0371, "grad_norm": 21.77640151977539, "learning_rate": 9.90909090909091e-06, "num_tokens": 60623.0, "completions/mean_length": 79.25, "completions/min_length": 61.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.25, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.3762175440788269, "rewards/meter/std": 0.3404117226600647, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.995638370513916, "rewards/repeat_soft/std": 0.0033319119829684496, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.25587037205696106, "rewards/total_composite/mean": 0.4848466217517853, "rewards/total_composite/std": 0.30984121561050415, "reward": 0.4848466217517853, "reward_std": 0.30984121561050415, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2795345187187195, "sampling/sampling_logp_difference/max": 3.1914567947387695, "sampling/importance_sampling_ratio/min": 0.041111934930086136, "sampling/importance_sampling_ratio/mean": 1.0121794939041138, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1605252847075462, "clip_ratio/low_mean": 0.05587913282215595, "clip_ratio/low_min": 0.05587913282215595, "clip_ratio/high_mean": 0.14557625632733107, "clip_ratio/high_max": 0.14557625632733107, "clip_ratio/region_mean": 0.20145538914948702, "reward_total_mean": 0.4848466217517853, "reward_meter_mean": 0.3762175440788269, "reward_meter_std": 0.3404117226600647, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.995638370513916, "reward_repeat_soft_std": 0.0033319119829684496, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.25587037205696106, "reward_total_composite_mean": 0.4848466217517853, "reward_total_composite_std": 0.30984121561050415} {"timestamp_utc": "2026-04-12T22:17:51Z", "mode": "train", "global_step": 32, "epoch": 0.0032144650929181316, "loss": 0.0928, "grad_norm": 28.076183319091797, "learning_rate": 9.906060606060607e-06, "num_tokens": 62100.0, "completions/mean_length": 29.625, "completions/min_length": 22.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.625, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.48696160316467285, "rewards/meter/std": 0.4289628863334656, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9554293155670166, "rewards/repeat_soft/std": 0.013228065334260464, "rewards/judge_quality/mean": 0.9200000166893005, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.740675687789917, "rewards/total_composite/std": 0.19279979169368744, "reward": 0.740675687789917, "reward_std": 0.19279977679252625, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24338388442993164, "sampling/sampling_logp_difference/max": 2.743772506713867, "sampling/importance_sampling_ratio/min": 0.06432721018791199, "sampling/importance_sampling_ratio/mean": 1.0007094144821167, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2935088127851486, "clip_ratio/low_mean": 0.11610591504722834, "clip_ratio/low_min": 0.11610591504722834, "clip_ratio/high_mean": 0.10445739328861237, "clip_ratio/high_max": 0.10445739328861237, "clip_ratio/region_mean": 0.2205633083358407, "reward_total_mean": 0.740675687789917, "reward_meter_mean": 0.48696160316467285, "reward_meter_std": 0.4289628863334656, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9554293155670166, "reward_repeat_soft_std": 0.013228065334260464, "reward_judge_quality_mean": 0.9200000166893005, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.740675687789917, "reward_total_composite_std": 0.19279979169368744} {"timestamp_utc": "2026-04-12T22:17:57Z", "mode": "train", "global_step": 33, "epoch": 0.0033149171270718232, "loss": 0.0988, "grad_norm": 13.971663475036621, "learning_rate": 9.903030303030305e-06, "num_tokens": 63936.0, "completions/mean_length": 65.5, "completions/min_length": 36.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.36027342081069946, "rewards/meter/std": 0.43213415145874023, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9996891021728516, "rewards/repeat_soft/std": 0.0006069060764275491, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.19078317284584045, "rewards/total_composite/mean": 0.5131999254226685, "rewards/total_composite/std": 0.2780666947364807, "reward": 0.5131999254226685, "reward_std": 0.2780666649341583, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19844399392604828, "sampling/sampling_logp_difference/max": 2.652165412902832, "sampling/importance_sampling_ratio/min": 0.07049839198589325, "sampling/importance_sampling_ratio/mean": 1.0268425941467285, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8619239330291748, "clip_ratio/low_mean": 0.07098640222102404, "clip_ratio/low_min": 0.07098640222102404, "clip_ratio/high_mean": 0.10280672460794449, "clip_ratio/high_max": 0.10280672460794449, "clip_ratio/region_mean": 0.17379312682896852, "reward_total_mean": 0.5131999254226685, "reward_meter_mean": 0.36027342081069946, "reward_meter_std": 0.43213415145874023, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9996891021728516, "reward_repeat_soft_std": 0.0006069060764275491, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.19078317284584045, "reward_total_composite_mean": 0.5131999254226685, "reward_total_composite_std": 0.2780666947364807} {"timestamp_utc": "2026-04-12T22:18:03Z", "mode": "train", "global_step": 34, "epoch": 0.003415369161225515, "loss": 0.1421, "grad_norm": 17.26323699951172, "learning_rate": 9.9e-06, "num_tokens": 65663.0, "completions/mean_length": 57.875, "completions/min_length": 38.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.875, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.6663380861282349, "rewards/meter/std": 0.3654095232486725, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9954054355621338, "rewards/repeat_soft/std": 0.006265631411224604, "rewards/judge_quality/mean": 0.5637500286102295, "rewards/judge_quality/std": 0.22012579441070557, "rewards/total_composite/mean": 0.7185176610946655, "rewards/total_composite/std": 0.2052834928035736, "reward": 0.7185176610946655, "reward_std": 0.2052834928035736, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2225874960422516, "sampling/sampling_logp_difference/max": 1.3435239791870117, "sampling/importance_sampling_ratio/min": 0.2609245479106903, "sampling/importance_sampling_ratio/mean": 1.0131149291992188, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8193220719695091, "clip_ratio/low_mean": 0.09162838384509087, "clip_ratio/low_min": 0.09162838384509087, "clip_ratio/high_mean": 0.12510230764746666, "clip_ratio/high_max": 0.12510230764746666, "clip_ratio/region_mean": 0.21673069149255753, "reward_total_mean": 0.7185176610946655, "reward_meter_mean": 0.6663380861282349, "reward_meter_std": 0.3654095232486725, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9954054355621338, "reward_repeat_soft_std": 0.006265631411224604, "reward_judge_quality_mean": 0.5637500286102295, "reward_judge_quality_std": 0.22012579441070557, "reward_total_composite_mean": 0.7185176610946655, "reward_total_composite_std": 0.2052834928035736} {"timestamp_utc": "2026-04-12T22:18:09Z", "mode": "train", "global_step": 35, "epoch": 0.0035158211953792064, "loss": 0.1592, "grad_norm": 31.29047203063965, "learning_rate": 9.896969696969699e-06, "num_tokens": 67253.0, "completions/mean_length": 33.75, "completions/min_length": 28.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.75, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.5175750255584717, "rewards/meter/std": 0.333164244890213, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9875191450119019, "rewards/repeat_soft/std": 0.016329128295183182, "rewards/judge_quality/mean": 0.5974999666213989, "rewards/judge_quality/std": 0.211643248796463, "rewards/total_composite/mean": 0.6191242337226868, "rewards/total_composite/std": 0.29068228602409363, "reward": 0.6191242337226868, "reward_std": 0.29068228602409363, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20600245893001556, "sampling/sampling_logp_difference/max": 2.2265100479125977, "sampling/importance_sampling_ratio/min": 0.1079043596982956, "sampling/importance_sampling_ratio/mean": 1.0101490020751953, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9016148820519447, "clip_ratio/low_mean": 0.06943389028310776, "clip_ratio/low_min": 0.06943389028310776, "clip_ratio/high_mean": 0.11182209756225348, "clip_ratio/high_max": 0.11182209756225348, "clip_ratio/region_mean": 0.18125598784536123, "reward_total_mean": 0.6191242337226868, "reward_meter_mean": 0.5175750255584717, "reward_meter_std": 0.333164244890213, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9875191450119019, "reward_repeat_soft_std": 0.016329128295183182, "reward_judge_quality_mean": 0.5974999666213989, "reward_judge_quality_std": 0.211643248796463, "reward_total_composite_mean": 0.6191242337226868, "reward_total_composite_std": 0.29068228602409363} {"timestamp_utc": "2026-04-12T22:18:17Z", "mode": "train", "global_step": 36, "epoch": 0.003616273229532898, "loss": -0.0159, "grad_norm": 13.74673080444336, "learning_rate": 9.893939393939395e-06, "num_tokens": 68856.0, "completions/mean_length": 51.375, "completions/min_length": 38.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.375, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.3229061961174011, "rewards/meter/std": 0.35688337683677673, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9967907667160034, "rewards/repeat_soft/std": 0.003882752498611808, "rewards/judge_quality/mean": 0.6887500286102295, "rewards/judge_quality/std": 0.21931307017803192, "rewards/total_composite/mean": 0.601611852645874, "rewards/total_composite/std": 0.19675835967063904, "reward": 0.601611852645874, "reward_std": 0.19675834476947784, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1898266226053238, "sampling/sampling_logp_difference/max": 1.7012615203857422, "sampling/importance_sampling_ratio/min": 0.1824532151222229, "sampling/importance_sampling_ratio/mean": 1.0139268636703491, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4139694049954414, "clip_ratio/low_mean": 0.10160249006003141, "clip_ratio/low_min": 0.10160249006003141, "clip_ratio/high_mean": 0.03077651560306549, "clip_ratio/high_max": 0.03077651560306549, "clip_ratio/region_mean": 0.1323790056630969, "reward_total_mean": 0.601611852645874, "reward_meter_mean": 0.3229061961174011, "reward_meter_std": 0.35688337683677673, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9967907667160034, "reward_repeat_soft_std": 0.003882752498611808, "reward_judge_quality_mean": 0.6887500286102295, "reward_judge_quality_std": 0.21931307017803192, "reward_total_composite_mean": 0.601611852645874, "reward_total_composite_std": 0.19675835967063904} {"timestamp_utc": "2026-04-12T22:18:23Z", "mode": "train", "global_step": 37, "epoch": 0.0037167252636865896, "loss": -0.0113, "grad_norm": 13.515311241149902, "learning_rate": 9.890909090909092e-06, "num_tokens": 70697.0, "completions/mean_length": 59.125, "completions/min_length": 47.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.125, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.765489935874939, "rewards/meter/std": 0.34022775292396545, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9889501929283142, "rewards/repeat_soft/std": 0.010796717368066311, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.21685661375522614, "rewards/total_composite/mean": 0.7452405691146851, "rewards/total_composite/std": 0.13115821778774261, "reward": 0.7452405691146851, "reward_std": 0.13115820288658142, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20520399510860443, "sampling/sampling_logp_difference/max": 1.5714797973632812, "sampling/importance_sampling_ratio/min": 0.20773755013942719, "sampling/importance_sampling_ratio/mean": 1.0194767713546753, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.044485777616501, "clip_ratio/low_mean": 0.09386534243822098, "clip_ratio/low_min": 0.09386534243822098, "clip_ratio/high_mean": 0.11555166356265545, "clip_ratio/high_max": 0.11555166356265545, "clip_ratio/region_mean": 0.20941700600087643, "reward_total_mean": 0.7452405691146851, "reward_meter_mean": 0.765489935874939, "reward_meter_std": 0.34022775292396545, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9889501929283142, "reward_repeat_soft_std": 0.010796717368066311, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.21685661375522614, "reward_total_composite_mean": 0.7452405691146851, "reward_total_composite_std": 0.13115821778774261} {"timestamp_utc": "2026-04-12T22:18:29Z", "mode": "train", "global_step": 38, "epoch": 0.0038171772978402812, "loss": 0.0265, "grad_norm": 11.196772575378418, "learning_rate": 9.887878787878789e-06, "num_tokens": 72445.0, "completions/mean_length": 65.5, "completions/min_length": 60.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.5, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.8452023267745972, "rewards/meter/std": 0.2854308485984802, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9985214471817017, "rewards/repeat_soft/std": 0.004181958269327879, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.14201989769935608, "rewards/total_composite/mean": 0.7820681929588318, "rewards/total_composite/std": 0.09378696978092194, "reward": 0.7820681929588318, "reward_std": 0.09378696978092194, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16852843761444092, "sampling/sampling_logp_difference/max": 1.3398032188415527, "sampling/importance_sampling_ratio/min": 0.2618972063064575, "sampling/importance_sampling_ratio/mean": 1.028748631477356, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.285915844142437, "clip_ratio/low_mean": 0.017307693138718605, "clip_ratio/low_min": 0.017307693138718605, "clip_ratio/high_mean": 0.14909545984119177, "clip_ratio/high_max": 0.14909545984119177, "clip_ratio/region_mean": 0.16640315297991037, "reward_total_mean": 0.7820681929588318, "reward_meter_mean": 0.8452023267745972, "reward_meter_std": 0.2854308485984802, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9985214471817017, "reward_repeat_soft_std": 0.004181958269327879, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.14201989769935608, "reward_total_composite_mean": 0.7820681929588318, "reward_total_composite_std": 0.09378696978092194} {"timestamp_utc": "2026-04-12T22:18:36Z", "mode": "train", "global_step": 39, "epoch": 0.003917629331993973, "loss": 0.0498, "grad_norm": 21.736717224121094, "learning_rate": 9.884848484848486e-06, "num_tokens": 74160.0, "completions/mean_length": 56.375, "completions/min_length": 43.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.375, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.8037891387939453, "rewards/meter/std": 0.25627055764198303, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9939441680908203, "rewards/repeat_soft/std": 0.007073741871863604, "rewards/judge_quality/mean": 0.6525000333786011, "rewards/judge_quality/std": 0.1348809152841568, "rewards/total_composite/mean": 0.8068495392799377, "rewards/total_composite/std": 0.14063221216201782, "reward": 0.8068495392799377, "reward_std": 0.14063221216201782, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18867532908916473, "sampling/sampling_logp_difference/max": 2.8137412071228027, "sampling/importance_sampling_ratio/min": 0.05998017266392708, "sampling/importance_sampling_ratio/mean": 0.9923123121261597, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6779676862061024, "clip_ratio/low_mean": 0.06264539621770382, "clip_ratio/low_min": 0.06264539621770382, "clip_ratio/high_mean": 0.10982005670666695, "clip_ratio/high_max": 0.10982005670666695, "clip_ratio/region_mean": 0.17246545292437077, "reward_total_mean": 0.8068495392799377, "reward_meter_mean": 0.8037891387939453, "reward_meter_std": 0.25627055764198303, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9939441680908203, "reward_repeat_soft_std": 0.007073741871863604, "reward_judge_quality_mean": 0.6525000333786011, "reward_judge_quality_std": 0.1348809152841568, "reward_total_composite_mean": 0.8068495392799377, "reward_total_composite_std": 0.14063221216201782} {"timestamp_utc": "2026-04-12T22:18:42Z", "mode": "train", "global_step": 40, "epoch": 0.004018081366147665, "loss": 0.1016, "grad_norm": 22.00394058227539, "learning_rate": 9.881818181818182e-06, "num_tokens": 75789.0, "completions/mean_length": 40.625, "completions/min_length": 27.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.625, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.8645680546760559, "rewards/meter/std": 0.33847635984420776, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.952668309211731, "rewards/repeat_soft/std": 0.06325025111436844, "rewards/judge_quality/mean": 0.9200000166893005, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.9103224277496338, "rewards/total_composite/std": 0.15087567269802094, "reward": 0.9103224277496338, "reward_std": 0.15087567269802094, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1935538351535797, "sampling/sampling_logp_difference/max": 2.373173952102661, "sampling/importance_sampling_ratio/min": 0.09318449348211288, "sampling/importance_sampling_ratio/mean": 1.0082100629806519, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0727061294019222, "clip_ratio/low_mean": 0.02604166604578495, "clip_ratio/low_min": 0.02604166604578495, "clip_ratio/high_mean": 0.16503243800252676, "clip_ratio/high_max": 0.16503243800252676, "clip_ratio/region_mean": 0.1910741040483117, "reward_total_mean": 0.9103224277496338, "reward_meter_mean": 0.8645680546760559, "reward_meter_std": 0.33847635984420776, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.952668309211731, "reward_repeat_soft_std": 0.06325025111436844, "reward_judge_quality_mean": 0.9200000166893005, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.9103224277496338, "reward_total_composite_std": 0.15087567269802094} {"timestamp_utc": "2026-04-12T22:18:51Z", "mode": "train", "global_step": 41, "epoch": 0.004118533400301356, "loss": -0.1876, "grad_norm": 12.826306343078613, "learning_rate": 9.87878787878788e-06, "num_tokens": 77469.0, "completions/mean_length": 50.0, "completions/min_length": 24.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.0, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.6660476922988892, "rewards/meter/std": 0.40160611271858215, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.997750997543335, "rewards/repeat_soft/std": 0.005148967728018761, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.12906257808208466, "rewards/total_composite/mean": 0.6337647438049316, "rewards/total_composite/std": 0.2952870726585388, "reward": 0.6337647438049316, "reward_std": 0.29528704285621643, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20901450514793396, "sampling/sampling_logp_difference/max": 1.7976255416870117, "sampling/importance_sampling_ratio/min": 0.16569185256958008, "sampling/importance_sampling_ratio/mean": 1.0211420059204102, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4476252645254135, "clip_ratio/low_mean": 0.09334935806691647, "clip_ratio/low_min": 0.09334935806691647, "clip_ratio/high_mean": 0.13066565804183483, "clip_ratio/high_max": 0.13066565804183483, "clip_ratio/region_mean": 0.2240150161087513, "reward_total_mean": 0.6337647438049316, "reward_meter_mean": 0.6660476922988892, "reward_meter_std": 0.40160611271858215, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.997750997543335, "reward_repeat_soft_std": 0.005148967728018761, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.12906257808208466, "reward_total_composite_mean": 0.6337647438049316, "reward_total_composite_std": 0.2952870726585388} {"timestamp_utc": "2026-04-12T22:18:57Z", "mode": "train", "global_step": 42, "epoch": 0.004218985434455048, "loss": 0.0467, "grad_norm": 24.63319969177246, "learning_rate": 9.875757575757576e-06, "num_tokens": 79094.0, "completions/mean_length": 54.125, "completions/min_length": 34.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.125, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.8649886250495911, "rewards/meter/std": 0.33145052194595337, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9943860769271851, "rewards/repeat_soft/std": 0.008962815627455711, "rewards/judge_quality/mean": 0.4437499940395355, "rewards/judge_quality/std": 0.2084595113992691, "rewards/total_composite/mean": 0.7718085050582886, "rewards/total_composite/std": 0.18098533153533936, "reward": 0.7718085050582886, "reward_std": 0.18098531663417816, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2050308734178543, "sampling/sampling_logp_difference/max": 1.4114322662353516, "sampling/importance_sampling_ratio/min": 0.24379386007785797, "sampling/importance_sampling_ratio/mean": 1.0256385803222656, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6599382311105728, "clip_ratio/low_mean": 0.0513223260641098, "clip_ratio/low_min": 0.0513223260641098, "clip_ratio/high_mean": 0.13139697071164846, "clip_ratio/high_max": 0.13139697071164846, "clip_ratio/region_mean": 0.18271929677575827, "reward_total_mean": 0.7718085050582886, "reward_meter_mean": 0.8649886250495911, "reward_meter_std": 0.33145052194595337, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9943860769271851, "reward_repeat_soft_std": 0.008962815627455711, "reward_judge_quality_mean": 0.4437499940395355, "reward_judge_quality_std": 0.2084595113992691, "reward_total_composite_mean": 0.7718085050582886, "reward_total_composite_std": 0.18098533153533936} {"timestamp_utc": "2026-04-12T22:19:04Z", "mode": "train", "global_step": 43, "epoch": 0.004319437468608739, "loss": 0.0083, "grad_norm": 13.690433502197266, "learning_rate": 9.872727272727274e-06, "num_tokens": 80983.0, "completions/mean_length": 61.125, "completions/min_length": 50.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.125, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.7919253706932068, "rewards/meter/std": 0.27126315236091614, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.984146773815155, "rewards/repeat_soft/std": 0.01696406863629818, "rewards/judge_quality/mean": 0.768750011920929, "rewards/judge_quality/std": 0.21695540845394135, "rewards/total_composite/mean": 0.8354060649871826, "rewards/total_composite/std": 0.10415403544902802, "reward": 0.8354060649871826, "reward_std": 0.10415403544902802, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15504302084445953, "sampling/sampling_logp_difference/max": 1.9836196899414062, "sampling/importance_sampling_ratio/min": 0.1375703662633896, "sampling/importance_sampling_ratio/mean": 1.0255910158157349, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.252104938030243, "clip_ratio/low_mean": 0.06915658712387085, "clip_ratio/low_min": 0.06915658712387085, "clip_ratio/high_mean": 0.07177211018279195, "clip_ratio/high_max": 0.07177211018279195, "clip_ratio/region_mean": 0.1409286973066628, "reward_total_mean": 0.8354060649871826, "reward_meter_mean": 0.7919253706932068, "reward_meter_std": 0.27126315236091614, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.984146773815155, "reward_repeat_soft_std": 0.01696406863629818, "reward_judge_quality_mean": 0.768750011920929, "reward_judge_quality_std": 0.21695540845394135, "reward_total_composite_mean": 0.8354060649871826, "reward_total_composite_std": 0.10415403544902802} {"timestamp_utc": "2026-04-12T22:19:11Z", "mode": "train", "global_step": 44, "epoch": 0.004419889502762431, "loss": 0.0866, "grad_norm": 8.773218154907227, "learning_rate": 9.869696969696971e-06, "num_tokens": 83437.0, "completions/mean_length": 126.75, "completions/min_length": 103.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.75, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.3680556118488312, "rewards/meter/std": 0.2608366906642914, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9973905086517334, "rewards/repeat_soft/std": 0.0024202882777899504, "rewards/judge_quality/mean": 0.5074999928474426, "rewards/judge_quality/std": 0.1642080694437027, "rewards/total_composite/mean": 0.5676140785217285, "rewards/total_composite/std": 0.15416587889194489, "reward": 0.5676140785217285, "reward_std": 0.1541658639907837, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19027352333068848, "sampling/sampling_logp_difference/max": 1.7497572898864746, "sampling/importance_sampling_ratio/min": 0.17381611466407776, "sampling/importance_sampling_ratio/mean": 1.0400882959365845, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.894649162888527, "clip_ratio/low_mean": 0.11630215495824814, "clip_ratio/low_min": 0.11630215495824814, "clip_ratio/high_mean": 0.07096002250909805, "clip_ratio/high_max": 0.07096002250909805, "clip_ratio/region_mean": 0.1872621774673462, "reward_total_mean": 0.5676140785217285, "reward_meter_mean": 0.3680556118488312, "reward_meter_std": 0.2608366906642914, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9973905086517334, "reward_repeat_soft_std": 0.0024202882777899504, "reward_judge_quality_mean": 0.5074999928474426, "reward_judge_quality_std": 0.1642080694437027, "reward_total_composite_mean": 0.5676140785217285, "reward_total_composite_std": 0.15416587889194489} {"timestamp_utc": "2026-04-12T22:19:17Z", "mode": "train", "global_step": 45, "epoch": 0.0045203415369161224, "loss": -0.0796, "grad_norm": 12.670366287231445, "learning_rate": 9.866666666666668e-06, "num_tokens": 85060.0, "completions/mean_length": 51.875, "completions/min_length": 35.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.875, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.8333461284637451, "rewards/meter/std": 0.27241066098213196, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9904763698577881, "rewards/repeat_soft/std": 0.016579801216721535, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.8438034057617188, "rewards/total_composite/std": 0.17045538127422333, "reward": 0.8438034057617188, "reward_std": 0.17045536637306213, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16479381918907166, "sampling/sampling_logp_difference/max": 2.0195093154907227, "sampling/importance_sampling_ratio/min": 0.13272058963775635, "sampling/importance_sampling_ratio/mean": 1.0289181470870972, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5834421515464783, "clip_ratio/low_mean": 0.09027308598160744, "clip_ratio/low_min": 0.09027308598160744, "clip_ratio/high_mean": 0.07856571674346924, "clip_ratio/high_max": 0.07856571674346924, "clip_ratio/region_mean": 0.16883880272507668, "reward_total_mean": 0.8438034057617188, "reward_meter_mean": 0.8333461284637451, "reward_meter_std": 0.27241066098213196, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9904763698577881, "reward_repeat_soft_std": 0.016579801216721535, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.8438034057617188, "reward_total_composite_std": 0.17045538127422333} {"timestamp_utc": "2026-04-12T22:19:23Z", "mode": "train", "global_step": 46, "epoch": 0.0046207935710698145, "loss": -0.0367, "grad_norm": 15.993819236755371, "learning_rate": 9.863636363636364e-06, "num_tokens": 86639.0, "completions/mean_length": 41.375, "completions/min_length": 34.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.375, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.6189440488815308, "rewards/meter/std": 0.4612226188182831, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9973520040512085, "rewards/repeat_soft/std": 0.00505313603207469, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404788017273, "rewards/total_composite/mean": 0.6951350569725037, "rewards/total_composite/std": 0.24396578967571259, "reward": 0.6951350569725037, "reward_std": 0.2439657747745514, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22641699016094208, "sampling/sampling_logp_difference/max": 1.6650819778442383, "sampling/importance_sampling_ratio/min": 0.18917514383792877, "sampling/importance_sampling_ratio/mean": 1.0454026460647583, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.159736454486847, "clip_ratio/low_mean": 0.0926593616604805, "clip_ratio/low_min": 0.0926593616604805, "clip_ratio/high_mean": 0.11545929498970509, "clip_ratio/high_max": 0.11545929498970509, "clip_ratio/region_mean": 0.20811865665018559, "reward_total_mean": 0.6951350569725037, "reward_meter_mean": 0.6189440488815308, "reward_meter_std": 0.4612226188182831, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9973520040512085, "reward_repeat_soft_std": 0.00505313603207469, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404788017273, "reward_total_composite_mean": 0.6951350569725037, "reward_total_composite_std": 0.24396578967571259} {"timestamp_utc": "2026-04-12T22:19:33Z", "mode": "train", "global_step": 47, "epoch": 0.004721245605223506, "loss": 0.1663, "grad_norm": 14.117936134338379, "learning_rate": 9.860606060606061e-06, "num_tokens": 88311.0, "completions/mean_length": 60.0, "completions/min_length": 33.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.5848500728607178, "rewards/meter/std": 0.310144305229187, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9949307441711426, "rewards/repeat_soft/std": 0.0036206496879458427, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.6949256062507629, "rewards/total_composite/std": 0.19530318677425385, "reward": 0.6949256062507629, "reward_std": 0.19530318677425385, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21896903216838837, "sampling/sampling_logp_difference/max": 1.9479656219482422, "sampling/importance_sampling_ratio/min": 0.1425638049840927, "sampling/importance_sampling_ratio/mean": 0.9914653301239014, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5602813884615898, "clip_ratio/low_mean": 0.0784417437389493, "clip_ratio/low_min": 0.0784417437389493, "clip_ratio/high_mean": 0.10587121546268463, "clip_ratio/high_max": 0.10587121546268463, "clip_ratio/region_mean": 0.18431295920163393, "reward_total_mean": 0.6949256062507629, "reward_meter_mean": 0.5848500728607178, "reward_meter_std": 0.310144305229187, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9949307441711426, "reward_repeat_soft_std": 0.0036206496879458427, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.6949256062507629, "reward_total_composite_std": 0.19530318677425385} {"timestamp_utc": "2026-04-12T22:19:39Z", "mode": "train", "global_step": 48, "epoch": 0.004821697639377198, "loss": 0.1596, "grad_norm": 25.422317504882812, "learning_rate": 9.857575757575758e-06, "num_tokens": 89968.0, "completions/mean_length": 38.125, "completions/min_length": 26.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.125, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.48131677508354187, "rewards/meter/std": 0.264057993888855, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9779799580574036, "rewards/repeat_soft/std": 0.028363212943077087, "rewards/judge_quality/mean": 0.6225000023841858, "rewards/judge_quality/std": 0.24656209349632263, "rewards/total_composite/mean": 0.651140570640564, "rewards/total_composite/std": 0.1632310152053833, "reward": 0.651140570640564, "reward_std": 0.1632310003042221, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24116353690624237, "sampling/sampling_logp_difference/max": 1.608734130859375, "sampling/importance_sampling_ratio/min": 0.2001408040523529, "sampling/importance_sampling_ratio/mean": 1.0072649717330933, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5285140573978424, "clip_ratio/low_mean": 0.13228978216648102, "clip_ratio/low_min": 0.13228978216648102, "clip_ratio/high_mean": 0.0881993044167757, "clip_ratio/high_max": 0.0881993044167757, "clip_ratio/region_mean": 0.22048908658325672, "reward_total_mean": 0.651140570640564, "reward_meter_mean": 0.48131677508354187, "reward_meter_std": 0.264057993888855, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9779799580574036, "reward_repeat_soft_std": 0.028363212943077087, "reward_judge_quality_mean": 0.6225000023841858, "reward_judge_quality_std": 0.24656209349632263, "reward_total_composite_mean": 0.651140570640564, "reward_total_composite_std": 0.1632310152053833} {"timestamp_utc": "2026-04-12T22:19:47Z", "mode": "train", "global_step": 49, "epoch": 0.004922149673530889, "loss": 0.0595, "grad_norm": 16.515077590942383, "learning_rate": 9.854545454545456e-06, "num_tokens": 92540.0, "completions/mean_length": 130.5, "completions/min_length": 119.0, "completions/max_length": 148.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 130.5, "completions/min_terminated_length": 119.0, "completions/max_terminated_length": 148.0, "rewards/meter/mean": 0.4894363284111023, "rewards/meter/std": 0.22659291326999664, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9929240345954895, "rewards/repeat_soft/std": 0.006053104996681213, "rewards/judge_quality/mean": 0.4987499713897705, "rewards/judge_quality/std": 0.13695022463798523, "rewards/total_composite/mean": 0.5284908413887024, "rewards/total_composite/std": 0.22537189722061157, "reward": 0.5284908413887024, "reward_std": 0.22537189722061157, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2464165985584259, "sampling/sampling_logp_difference/max": 3.279010057449341, "sampling/importance_sampling_ratio/min": 0.03766552358865738, "sampling/importance_sampling_ratio/mean": 0.9850641489028931, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1143654063344002, "clip_ratio/low_mean": 0.050971951335668564, "clip_ratio/low_min": 0.050971951335668564, "clip_ratio/high_mean": 0.16969024203717709, "clip_ratio/high_max": 0.16969024203717709, "clip_ratio/region_mean": 0.22066219337284565, "reward_total_mean": 0.5284908413887024, "reward_meter_mean": 0.4894363284111023, "reward_meter_std": 0.22659291326999664, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9929240345954895, "reward_repeat_soft_std": 0.006053104996681213, "reward_judge_quality_mean": 0.4987499713897705, "reward_judge_quality_std": 0.13695022463798523, "reward_total_composite_mean": 0.5284908413887024, "reward_total_composite_std": 0.22537189722061157} {"timestamp_utc": "2026-04-12T22:19:54Z", "mode": "train", "global_step": 50, "epoch": 0.005022601707684581, "loss": 0.1871, "grad_norm": 12.148331642150879, "learning_rate": 9.851515151515151e-06, "num_tokens": 94611.0, "completions/mean_length": 96.875, "completions/min_length": 60.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.875, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.33955442905426025, "rewards/meter/std": 0.31331250071525574, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9961608648300171, "rewards/repeat_soft/std": 0.005006108433008194, "rewards/judge_quality/mean": 0.3712499737739563, "rewards/judge_quality/std": 0.21390503644943237, "rewards/total_composite/mean": 0.5137906074523926, "rewards/total_composite/std": 0.15746544301509857, "reward": 0.5137906074523926, "reward_std": 0.15746545791625977, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2518634796142578, "sampling/sampling_logp_difference/max": 2.144465446472168, "sampling/importance_sampling_ratio/min": 0.11713062971830368, "sampling/importance_sampling_ratio/mean": 1.0296486616134644, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0926555544137955, "clip_ratio/low_mean": 0.13645275123417377, "clip_ratio/low_min": 0.13645275123417377, "clip_ratio/high_mean": 0.07529207319021225, "clip_ratio/high_max": 0.07529207319021225, "clip_ratio/region_mean": 0.21174482442438602, "reward_total_mean": 0.5137906074523926, "reward_meter_mean": 0.33955442905426025, "reward_meter_std": 0.31331250071525574, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9961608648300171, "reward_repeat_soft_std": 0.005006108433008194, "reward_judge_quality_mean": 0.3712499737739563, "reward_judge_quality_std": 0.21390503644943237, "reward_total_composite_mean": 0.5137906074523926, "reward_total_composite_std": 0.15746544301509857} {"timestamp_utc": "2026-04-12T22:20:44Z", "mode": "eval", "global_step": 50, "epoch": 0.005022601707684581, "eval_loss": NaN, "eval_runtime": 50.3639, "eval_samples_per_second": 1.588, "eval_steps_per_second": 0.199, "eval_num_tokens": 94611.0, "eval_completions/mean_length": 92.5125, "eval_completions/min_length": 36.6, "eval_completions/max_length": 172.9, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 92.5125, "eval_completions/min_terminated_length": 36.6, "eval_completions/max_terminated_length": 172.9, "eval_rewards/meter/mean": 0.622799813747406, "eval_rewards/meter/std": 0.385142882168293, "eval_rewards/count_adherence/mean": 0.9683333277702332, "eval_rewards/count_adherence/std": 0.07128634452819824, "eval_rewards/hard_gate/mean": 0.95, "eval_rewards/hard_gate/std": 0.1414213538169861, "eval_rewards/repeat_soft/mean": 0.993061774969101, "eval_rewards/repeat_soft/std": 0.011147955048363656, "eval_rewards/judge_quality/mean": 0.4986249953508377, "eval_rewards/judge_quality/std": 0.16713942140340804, "eval_rewards/total_composite/mean": 0.6409295320510864, "eval_rewards/total_composite/std": 0.23319732695817946, "eval_reward": 0.6409295320510864, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.12240503132343292, "eval_sampling/sampling_logp_difference/max": 1.1358956813812255, "eval_sampling/importance_sampling_ratio/min": 0.32323764860630033, "eval_sampling/importance_sampling_ratio/mean": 1.031867229938507, "eval_sampling/importance_sampling_ratio/max": 1.530899715423584, "eval_entropy": 1.7908539295196533, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6409295320510864, "eval_reward_meter_mean": 0.622799813747406, "eval_reward_meter_std": 0.385142882168293, "eval_reward_count_adherence_mean": 0.9683333277702332, "eval_reward_count_adherence_std": 0.07128634452819824, "eval_reward_hard_gate_mean": 0.95, "eval_reward_hard_gate_std": 0.1414213538169861, "eval_reward_repeat_soft_mean": 0.993061774969101, "eval_reward_repeat_soft_std": 0.011147955048363656, "eval_reward_judge_quality_mean": 0.4986249953508377, "eval_reward_judge_quality_std": 0.16713942140340804, "eval_reward_total_composite_mean": 0.6409295320510864, "eval_reward_total_composite_std": 0.23319732695817946} {"timestamp_utc": "2026-04-12T22:20:53Z", "mode": "train", "global_step": 51, "epoch": 0.005123053741838272, "loss": 0.102, "grad_norm": 15.290033340454102, "learning_rate": 9.84848484848485e-06, "num_tokens": 96345.0, "completions/mean_length": 49.75, "completions/min_length": 35.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.75, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.8911745548248291, "rewards/meter/std": 0.18275241553783417, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9963257312774658, "rewards/repeat_soft/std": 0.005425546783953905, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.8152861595153809, "rewards/total_composite/std": 0.12008143216371536, "reward": 0.8152861595153809, "reward_std": 0.12008143216371536, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2254175841808319, "sampling/sampling_logp_difference/max": 2.007718324661255, "sampling/importance_sampling_ratio/min": 0.13429473340511322, "sampling/importance_sampling_ratio/mean": 1.0337834358215332, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1199056655168533, "clip_ratio/low_mean": 0.07456336915493011, "clip_ratio/low_min": 0.07456336915493011, "clip_ratio/high_mean": 0.1384337618947029, "clip_ratio/high_max": 0.1384337618947029, "clip_ratio/region_mean": 0.21299713104963303, "reward_total_mean": 0.8152861595153809, "reward_meter_mean": 0.8911745548248291, "reward_meter_std": 0.18275241553783417, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9963257312774658, "reward_repeat_soft_std": 0.005425546783953905, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.8152861595153809, "reward_total_composite_std": 0.12008143216371536} {"timestamp_utc": "2026-04-12T22:21:00Z", "mode": "train", "global_step": 52, "epoch": 0.005223505775991964, "loss": 0.1486, "grad_norm": 10.75352668762207, "learning_rate": 9.845454545454546e-06, "num_tokens": 98656.0, "completions/mean_length": 114.875, "completions/min_length": 73.0, "completions/max_length": 159.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.875, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 159.0, "rewards/meter/mean": 0.7699568867683411, "rewards/meter/std": 0.3357446789741516, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9915257692337036, "rewards/repeat_soft/std": 0.004899078514426947, "rewards/judge_quality/mean": 0.6325000524520874, "rewards/judge_quality/std": 0.18850921094417572, "rewards/total_composite/mean": 0.7806956768035889, "rewards/total_composite/std": 0.20309355854988098, "reward": 0.7806956768035889, "reward_std": 0.20309355854988098, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18153929710388184, "sampling/sampling_logp_difference/max": 1.4434322118759155, "sampling/importance_sampling_ratio/min": 0.2361159771680832, "sampling/importance_sampling_ratio/mean": 1.042791724205017, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0450744926929474, "clip_ratio/low_mean": 0.051473021507263184, "clip_ratio/low_min": 0.051473021507263184, "clip_ratio/high_mean": 0.1433755587786436, "clip_ratio/high_max": 0.1433755587786436, "clip_ratio/region_mean": 0.1948485802859068, "reward_total_mean": 0.7806956768035889, "reward_meter_mean": 0.7699568867683411, "reward_meter_std": 0.3357446789741516, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9915257692337036, "reward_repeat_soft_std": 0.004899078514426947, "reward_judge_quality_mean": 0.6325000524520874, "reward_judge_quality_std": 0.18850921094417572, "reward_total_composite_mean": 0.7806956768035889, "reward_total_composite_std": 0.20309355854988098} {"timestamp_utc": "2026-04-12T22:21:06Z", "mode": "train", "global_step": 53, "epoch": 0.005323957810145655, "loss": -0.0155, "grad_norm": 24.79887580871582, "learning_rate": 9.842424242424243e-06, "num_tokens": 100184.0, "completions/mean_length": 38.0, "completions/min_length": 28.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.8915675282478333, "rewards/meter/std": 0.14790304005146027, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9969451427459717, "rewards/repeat_soft/std": 0.002551683457568288, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.7729640007019043, "rewards/total_composite/std": 0.33475446701049805, "reward": 0.7729640007019043, "reward_std": 0.33475446701049805, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21401749551296234, "sampling/sampling_logp_difference/max": 2.617398500442505, "sampling/importance_sampling_ratio/min": 0.07299250364303589, "sampling/importance_sampling_ratio/mean": 0.976485013961792, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6175712272524834, "clip_ratio/low_mean": 0.037664955481886864, "clip_ratio/low_min": 0.037664955481886864, "clip_ratio/high_mean": 0.09748073294758797, "clip_ratio/high_max": 0.09748073294758797, "clip_ratio/region_mean": 0.13514568842947483, "reward_total_mean": 0.7729640007019043, "reward_meter_mean": 0.8915675282478333, "reward_meter_std": 0.14790304005146027, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9969451427459717, "reward_repeat_soft_std": 0.002551683457568288, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.7729640007019043, "reward_total_composite_std": 0.33475446701049805} {"timestamp_utc": "2026-04-12T22:21:15Z", "mode": "train", "global_step": 54, "epoch": 0.005424409844299347, "loss": 0.0579, "grad_norm": 10.786398887634277, "learning_rate": 9.83939393939394e-06, "num_tokens": 102660.0, "completions/mean_length": 132.5, "completions/min_length": 94.0, "completions/max_length": 197.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.5, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 197.0, "rewards/meter/mean": 0.5160332918167114, "rewards/meter/std": 0.3664347231388092, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9905648231506348, "rewards/repeat_soft/std": 0.006998918484896421, "rewards/judge_quality/mean": 0.8237500190734863, "rewards/judge_quality/std": 0.2722361385822296, "rewards/total_composite/mean": 0.5760136842727661, "rewards/total_composite/std": 0.37488675117492676, "reward": 0.5760136842727661, "reward_std": 0.37488675117492676, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21526861190795898, "sampling/sampling_logp_difference/max": 2.1233134269714355, "sampling/importance_sampling_ratio/min": 0.11963456869125366, "sampling/importance_sampling_ratio/mean": 1.02375066280365, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.026296839118004, "clip_ratio/low_mean": 0.06303927302360535, "clip_ratio/low_min": 0.06303927302360535, "clip_ratio/high_mean": 0.13252770714461803, "clip_ratio/high_max": 0.13252770714461803, "clip_ratio/region_mean": 0.19556698016822338, "reward_total_mean": 0.5760136842727661, "reward_meter_mean": 0.5160332918167114, "reward_meter_std": 0.3664347231388092, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9905648231506348, "reward_repeat_soft_std": 0.006998918484896421, "reward_judge_quality_mean": 0.8237500190734863, "reward_judge_quality_std": 0.2722361385822296, "reward_total_composite_mean": 0.5760136842727661, "reward_total_composite_std": 0.37488675117492676} {"timestamp_utc": "2026-04-12T22:21:25Z", "mode": "train", "global_step": 55, "epoch": 0.0055248618784530384, "loss": 0.09, "grad_norm": 20.380891799926758, "learning_rate": 9.836363636363637e-06, "num_tokens": 104225.0, "completions/mean_length": 28.625, "completions/min_length": 24.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.625, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.6425614356994629, "rewards/meter/std": 0.46263325214385986, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9575977325439453, "rewards/repeat_soft/std": 0.013865554705262184, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.7381623983383179, "rewards/total_composite/std": 0.23571598529815674, "reward": 0.7381623983383179, "reward_std": 0.23571597039699554, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22299328446388245, "sampling/sampling_logp_difference/max": 1.5237703323364258, "sampling/importance_sampling_ratio/min": 0.21788883209228516, "sampling/importance_sampling_ratio/mean": 1.0329948663711548, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6725586950778961, "clip_ratio/low_mean": 0.0873684212565422, "clip_ratio/low_min": 0.0873684212565422, "clip_ratio/high_mean": 0.12113425973802805, "clip_ratio/high_max": 0.12113425973802805, "clip_ratio/region_mean": 0.20850268099457026, "reward_total_mean": 0.7381623983383179, "reward_meter_mean": 0.6425614356994629, "reward_meter_std": 0.46263325214385986, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9575977325439453, "reward_repeat_soft_std": 0.013865554705262184, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.7381623983383179, "reward_total_composite_std": 0.23571598529815674} {"timestamp_utc": "2026-04-12T22:21:37Z", "mode": "train", "global_step": 56, "epoch": 0.0056253139126067305, "loss": 0.044, "grad_norm": 17.16199493408203, "learning_rate": 9.833333333333333e-06, "num_tokens": 105784.0, "completions/mean_length": 36.875, "completions/min_length": 27.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.875, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.6049321889877319, "rewards/meter/std": 0.4337708652019501, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.6489694714546204, "rewards/total_composite/std": 0.19477401673793793, "reward": 0.6489694714546204, "reward_std": 0.19477401673793793, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18272539973258972, "sampling/sampling_logp_difference/max": 1.233874797821045, "sampling/importance_sampling_ratio/min": 0.2911621928215027, "sampling/importance_sampling_ratio/mean": 1.0084818601608276, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6532663255929947, "clip_ratio/low_mean": 0.10378074645996094, "clip_ratio/low_min": 0.10378074645996094, "clip_ratio/high_mean": 0.11709216423332691, "clip_ratio/high_max": 0.11709216423332691, "clip_ratio/region_mean": 0.22087291069328785, "reward_total_mean": 0.6489694714546204, "reward_meter_mean": 0.6049321889877319, "reward_meter_std": 0.4337708652019501, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.6489694714546204, "reward_total_composite_std": 0.19477401673793793} {"timestamp_utc": "2026-04-12T22:21:47Z", "mode": "train", "global_step": 57, "epoch": 0.005725765946760422, "loss": 0.1098, "grad_norm": 9.582910537719727, "learning_rate": 9.830303030303032e-06, "num_tokens": 108140.0, "completions/mean_length": 124.5, "completions/min_length": 98.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 124.5, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.6479657888412476, "rewards/meter/std": 0.37889355421066284, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9963884353637695, "rewards/repeat_soft/std": 0.003934774082154036, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.6608484387397766, "rewards/total_composite/std": 0.17602038383483887, "reward": 0.6608484387397766, "reward_std": 0.17602038383483887, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19246143102645874, "sampling/sampling_logp_difference/max": 2.835052490234375, "sampling/importance_sampling_ratio/min": 0.05871544033288956, "sampling/importance_sampling_ratio/mean": 1.0382243394851685, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9649572968482971, "clip_ratio/low_mean": 0.08068174216896296, "clip_ratio/low_min": 0.08068174216896296, "clip_ratio/high_mean": 0.1007919292896986, "clip_ratio/high_max": 0.1007919292896986, "clip_ratio/region_mean": 0.18147367145866156, "reward_total_mean": 0.6608484387397766, "reward_meter_mean": 0.6479657888412476, "reward_meter_std": 0.37889355421066284, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9963884353637695, "reward_repeat_soft_std": 0.003934774082154036, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.6608484387397766, "reward_total_composite_std": 0.17602038383483887} {"timestamp_utc": "2026-04-12T22:21:58Z", "mode": "train", "global_step": 58, "epoch": 0.005826217980914114, "loss": 0.0677, "grad_norm": 10.142060279846191, "learning_rate": 9.827272727272729e-06, "num_tokens": 110448.0, "completions/mean_length": 105.5, "completions/min_length": 94.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.5, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.6721632480621338, "rewards/meter/std": 0.34179237484931946, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9995437860488892, "rewards/repeat_soft/std": 0.0007127728313207626, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.6385196447372437, "rewards/total_composite/std": 0.3149482011795044, "reward": 0.6385196447372437, "reward_std": 0.3149482011795044, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21328410506248474, "sampling/sampling_logp_difference/max": 1.8783636093139648, "sampling/importance_sampling_ratio/min": 0.1528400182723999, "sampling/importance_sampling_ratio/mean": 1.0194169282913208, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9980643540620804, "clip_ratio/low_mean": 0.09210692159831524, "clip_ratio/low_min": 0.09210692159831524, "clip_ratio/high_mean": 0.09094057232141495, "clip_ratio/high_max": 0.09094057232141495, "clip_ratio/region_mean": 0.1830474939197302, "reward_total_mean": 0.6385196447372437, "reward_meter_mean": 0.6721632480621338, "reward_meter_std": 0.34179237484931946, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9995437860488892, "reward_repeat_soft_std": 0.0007127728313207626, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.6385196447372437, "reward_total_composite_std": 0.3149482011795044} {"timestamp_utc": "2026-04-12T22:22:06Z", "mode": "train", "global_step": 59, "epoch": 0.005926670015067805, "loss": -0.021, "grad_norm": 14.809919357299805, "learning_rate": 9.824242424242425e-06, "num_tokens": 112380.0, "completions/mean_length": 62.5, "completions/min_length": 54.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.5, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.6479412317276001, "rewards/meter/std": 0.38575154542922974, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9950402975082397, "rewards/repeat_soft/std": 0.008863254450261593, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.2670440077781677, "rewards/total_composite/mean": 0.6347871422767639, "rewards/total_composite/std": 0.33347341418266296, "reward": 0.6347871422767639, "reward_std": 0.3334733843803406, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2120247334241867, "sampling/sampling_logp_difference/max": 1.6805639266967773, "sampling/importance_sampling_ratio/min": 0.1862688958644867, "sampling/importance_sampling_ratio/mean": 1.0009524822235107, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7054464370012283, "clip_ratio/low_mean": 0.05873700138181448, "clip_ratio/low_min": 0.05873700138181448, "clip_ratio/high_mean": 0.13117719814181328, "clip_ratio/high_max": 0.13117719814181328, "clip_ratio/region_mean": 0.18991419952362776, "reward_total_mean": 0.6347871422767639, "reward_meter_mean": 0.6479412317276001, "reward_meter_std": 0.38575154542922974, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9950402975082397, "reward_repeat_soft_std": 0.008863254450261593, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.2670440077781677, "reward_total_composite_mean": 0.6347871422767639, "reward_total_composite_std": 0.33347341418266296} {"timestamp_utc": "2026-04-12T22:22:14Z", "mode": "train", "global_step": 60, "epoch": 0.006027122049221497, "loss": 0.1206, "grad_norm": 14.03058910369873, "learning_rate": 9.821212121212122e-06, "num_tokens": 114047.0, "completions/mean_length": 63.375, "completions/min_length": 45.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.375, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.682119607925415, "rewards/meter/std": 0.40137389302253723, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9977132081985474, "rewards/repeat_soft/std": 0.00379719166085124, "rewards/judge_quality/mean": 0.7699999809265137, "rewards/judge_quality/std": 0.22677870094776154, "rewards/total_composite/mean": 0.7877250909805298, "rewards/total_composite/std": 0.19343271851539612, "reward": 0.7877250909805298, "reward_std": 0.19343270361423492, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20363926887512207, "sampling/sampling_logp_difference/max": 1.5578479766845703, "sampling/importance_sampling_ratio/min": 0.21058876812458038, "sampling/importance_sampling_ratio/mean": 1.0375148057937622, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7059862613677979, "clip_ratio/low_mean": 0.0698032584041357, "clip_ratio/low_min": 0.0698032584041357, "clip_ratio/high_mean": 0.09936619736254215, "clip_ratio/high_max": 0.09936619736254215, "clip_ratio/region_mean": 0.16916945576667786, "reward_total_mean": 0.7877250909805298, "reward_meter_mean": 0.682119607925415, "reward_meter_std": 0.40137389302253723, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9977132081985474, "reward_repeat_soft_std": 0.00379719166085124, "reward_judge_quality_mean": 0.7699999809265137, "reward_judge_quality_std": 0.22677870094776154, "reward_total_composite_mean": 0.7877250909805298, "reward_total_composite_std": 0.19343271851539612} {"timestamp_utc": "2026-04-12T22:22:21Z", "mode": "train", "global_step": 61, "epoch": 0.006127574083375188, "loss": -0.0091, "grad_norm": 13.446069717407227, "learning_rate": 9.81818181818182e-06, "num_tokens": 115837.0, "completions/mean_length": 49.75, "completions/min_length": 32.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.75, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.4147126078605652, "rewards/meter/std": 0.41163748502731323, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9938614368438721, "rewards/repeat_soft/std": 0.010955816134810448, "rewards/judge_quality/mean": 0.5012500286102295, "rewards/judge_quality/std": 0.16974246501922607, "rewards/total_composite/mean": 0.5863817930221558, "rewards/total_composite/std": 0.2178468108177185, "reward": 0.5863817930221558, "reward_std": 0.2178467959165573, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19411690533161163, "sampling/sampling_logp_difference/max": 1.7225580215454102, "sampling/importance_sampling_ratio/min": 0.17860867083072662, "sampling/importance_sampling_ratio/mean": 1.0362242460250854, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.983920469880104, "clip_ratio/low_mean": 0.1204979931935668, "clip_ratio/low_min": 0.1204979931935668, "clip_ratio/high_mean": 0.0703886691480875, "clip_ratio/high_max": 0.0703886691480875, "clip_ratio/region_mean": 0.1908866623416543, "reward_total_mean": 0.5863817930221558, "reward_meter_mean": 0.4147126078605652, "reward_meter_std": 0.41163748502731323, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9938614368438721, "reward_repeat_soft_std": 0.010955816134810448, "reward_judge_quality_mean": 0.5012500286102295, "reward_judge_quality_std": 0.16974246501922607, "reward_total_composite_mean": 0.5863817930221558, "reward_total_composite_std": 0.2178468108177185} {"timestamp_utc": "2026-04-12T22:22:30Z", "mode": "train", "global_step": 62, "epoch": 0.00622802611752888, "loss": 0.0554, "grad_norm": 8.62053108215332, "learning_rate": 9.815151515151516e-06, "num_tokens": 118853.0, "completions/mean_length": 173.0, "completions/min_length": 155.0, "completions/max_length": 207.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 173.0, "completions/min_terminated_length": 155.0, "completions/max_terminated_length": 207.0, "rewards/meter/mean": 0.4597916603088379, "rewards/meter/std": 0.29323315620422363, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.99494469165802, "rewards/repeat_soft/std": 0.005377884954214096, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5711507201194763, "rewards/total_composite/std": 0.12056976556777954, "reward": 0.5711507201194763, "reward_std": 0.12056976556777954, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20841319859027863, "sampling/sampling_logp_difference/max": 2.5847525596618652, "sampling/importance_sampling_ratio/min": 0.07541473954916, "sampling/importance_sampling_ratio/mean": 1.0249940156936646, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5373004227876663, "clip_ratio/low_mean": 0.09918828681111336, "clip_ratio/low_min": 0.09918828681111336, "clip_ratio/high_mean": 0.11019240505993366, "clip_ratio/high_max": 0.11019240505993366, "clip_ratio/region_mean": 0.20938069187104702, "reward_total_mean": 0.5711507201194763, "reward_meter_mean": 0.4597916603088379, "reward_meter_std": 0.29323315620422363, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.99494469165802, "reward_repeat_soft_std": 0.005377884954214096, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5711507201194763, "reward_total_composite_std": 0.12056976556777954} {"timestamp_utc": "2026-04-12T22:22:37Z", "mode": "train", "global_step": 63, "epoch": 0.006328478151682571, "loss": 0.2431, "grad_norm": 20.96488380432129, "learning_rate": 9.812121212121212e-06, "num_tokens": 120739.0, "completions/mean_length": 74.75, "completions/min_length": 57.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.75, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.45688575506210327, "rewards/meter/std": 0.35617753863334656, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9985731840133667, "rewards/repeat_soft/std": 0.0012093555415049195, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.6039559245109558, "rewards/total_composite/std": 0.18145155906677246, "reward": 0.6039559245109558, "reward_std": 0.18145155906677246, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21291305124759674, "sampling/sampling_logp_difference/max": 2.4753305912017822, "sampling/importance_sampling_ratio/min": 0.08413517475128174, "sampling/importance_sampling_ratio/mean": 1.0076979398727417, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8964705485850573, "clip_ratio/low_mean": 0.128117217682302, "clip_ratio/low_min": 0.128117217682302, "clip_ratio/high_mean": 0.04571073269471526, "clip_ratio/high_max": 0.04571073269471526, "clip_ratio/region_mean": 0.17382795037701726, "reward_total_mean": 0.6039559245109558, "reward_meter_mean": 0.45688575506210327, "reward_meter_std": 0.35617753863334656, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9985731840133667, "reward_repeat_soft_std": 0.0012093555415049195, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.6039559245109558, "reward_total_composite_std": 0.18145155906677246} {"timestamp_utc": "2026-04-12T22:22:49Z", "mode": "train", "global_step": 64, "epoch": 0.006428930185836263, "loss": 0.0561, "grad_norm": 11.986035346984863, "learning_rate": 9.809090909090911e-06, "num_tokens": 123013.0, "completions/mean_length": 102.25, "completions/min_length": 84.0, "completions/max_length": 119.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.25, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.2554526627063751, "rewards/meter/std": 0.23822079598903656, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9972249269485474, "rewards/repeat_soft/std": 0.0024571451358497143, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.43968093395233154, "rewards/total_composite/std": 0.20654380321502686, "reward": 0.43968093395233154, "reward_std": 0.20654380321502686, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23794835805892944, "sampling/sampling_logp_difference/max": 1.656651258468628, "sampling/importance_sampling_ratio/min": 0.1907767653465271, "sampling/importance_sampling_ratio/mean": 1.025471806526184, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1588610261678696, "clip_ratio/low_mean": 0.10111558996140957, "clip_ratio/low_min": 0.10111558996140957, "clip_ratio/high_mean": 0.1057747658342123, "clip_ratio/high_max": 0.1057747658342123, "clip_ratio/region_mean": 0.20689035579562187, "reward_total_mean": 0.43968093395233154, "reward_meter_mean": 0.2554526627063751, "reward_meter_std": 0.23822079598903656, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9972249269485474, "reward_repeat_soft_std": 0.0024571451358497143, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.43968093395233154, "reward_total_composite_std": 0.20654380321502686} {"timestamp_utc": "2026-04-12T22:22:56Z", "mode": "train", "global_step": 65, "epoch": 0.0065293822199899544, "loss": 0.0258, "grad_norm": 21.1607723236084, "learning_rate": 9.806060606060607e-06, "num_tokens": 124597.0, "completions/mean_length": 48.0, "completions/min_length": 40.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.0, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.8362926840782166, "rewards/meter/std": 0.15002521872520447, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9975625872612, "rewards/repeat_soft/std": 0.004202974960207939, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.8105879426002502, "rewards/total_composite/std": 0.09124923497438431, "reward": 0.8105879426002502, "reward_std": 0.09124922752380371, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19722096621990204, "sampling/sampling_logp_difference/max": 4.355945110321045, "sampling/importance_sampling_ratio/min": 0.012830307707190514, "sampling/importance_sampling_ratio/mean": 1.0022335052490234, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8046837113797665, "clip_ratio/low_mean": 0.11510978080332279, "clip_ratio/low_min": 0.11510978080332279, "clip_ratio/high_mean": 0.06482380721718073, "clip_ratio/high_max": 0.06482380721718073, "clip_ratio/region_mean": 0.17993358802050352, "reward_total_mean": 0.8105879426002502, "reward_meter_mean": 0.8362926840782166, "reward_meter_std": 0.15002521872520447, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9975625872612, "reward_repeat_soft_std": 0.004202974960207939, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.8105879426002502, "reward_total_composite_std": 0.09124923497438431} {"timestamp_utc": "2026-04-12T22:23:09Z", "mode": "train", "global_step": 66, "epoch": 0.0066298342541436465, "loss": 0.7836, "grad_norm": 12.42125129699707, "learning_rate": 9.803030303030304e-06, "num_tokens": 126598.0, "completions/mean_length": 106.125, "completions/min_length": 35.0, "completions/max_length": 435.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.125, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 435.0, "rewards/meter/mean": 0.7745488882064819, "rewards/meter/std": 0.28584718704223633, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9980877637863159, "rewards/repeat_soft/std": 0.003999420441687107, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.21357084810733795, "rewards/total_composite/mean": 0.7292307615280151, "rewards/total_composite/std": 0.18889670073986053, "reward": 0.7292307615280151, "reward_std": 0.18889670073986053, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19706401228904724, "sampling/sampling_logp_difference/max": 1.9290742874145508, "sampling/importance_sampling_ratio/min": 0.14528262615203857, "sampling/importance_sampling_ratio/mean": 1.0292773246765137, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0778553932905197, "clip_ratio/low_mean": 0.0641922689974308, "clip_ratio/low_min": 0.0641922689974308, "clip_ratio/high_mean": 0.12423842307180166, "clip_ratio/high_max": 0.12423842307180166, "clip_ratio/region_mean": 0.18843069206923246, "reward_total_mean": 0.7292307615280151, "reward_meter_mean": 0.7745488882064819, "reward_meter_std": 0.28584718704223633, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9980877637863159, "reward_repeat_soft_std": 0.003999420441687107, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.21357084810733795, "reward_total_composite_mean": 0.7292307615280151, "reward_total_composite_std": 0.18889670073986053} {"timestamp_utc": "2026-04-12T22:23:17Z", "mode": "train", "global_step": 67, "epoch": 0.006730286288297338, "loss": -0.0138, "grad_norm": 12.676423072814941, "learning_rate": 9.800000000000001e-06, "num_tokens": 128362.0, "completions/mean_length": 61.5, "completions/min_length": 57.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.5, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9230592250823975, "rewards/meter/std": 0.17033740878105164, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.996062159538269, "rewards/repeat_soft/std": 0.007730784825980663, "rewards/judge_quality/mean": 0.5049999952316284, "rewards/judge_quality/std": 0.16801361739635468, "rewards/total_composite/mean": 0.8164828419685364, "rewards/total_composite/std": 0.10062293708324432, "reward": 0.8164828419685364, "reward_std": 0.10062292963266373, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16694454848766327, "sampling/sampling_logp_difference/max": 1.8855005502700806, "sampling/importance_sampling_ratio/min": 0.15175308287143707, "sampling/importance_sampling_ratio/mean": 1.0206559896469116, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.132050782442093, "clip_ratio/low_mean": 0.023706896230578423, "clip_ratio/low_min": 0.023706896230578423, "clip_ratio/high_mean": 0.15569907892495394, "clip_ratio/high_max": 0.15569907892495394, "clip_ratio/region_mean": 0.17940597515553236, "reward_total_mean": 0.8164828419685364, "reward_meter_mean": 0.9230592250823975, "reward_meter_std": 0.17033740878105164, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.996062159538269, "reward_repeat_soft_std": 0.007730784825980663, "reward_judge_quality_mean": 0.5049999952316284, "reward_judge_quality_std": 0.16801361739635468, "reward_total_composite_mean": 0.8164828419685364, "reward_total_composite_std": 0.10062293708324432} {"timestamp_utc": "2026-04-12T22:23:24Z", "mode": "train", "global_step": 68, "epoch": 0.00683073832245103, "loss": -0.0557, "grad_norm": 17.533666610717773, "learning_rate": 9.796969696969698e-06, "num_tokens": 129994.0, "completions/mean_length": 54.0, "completions/min_length": 37.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.29206711053848267, "rewards/meter/std": 0.2812570035457611, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9896067380905151, "rewards/repeat_soft/std": 0.01452327985316515, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.4468875229358673, "rewards/total_composite/std": 0.21935208141803741, "reward": 0.4468875229358673, "reward_std": 0.21935206651687622, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24508477747440338, "sampling/sampling_logp_difference/max": 2.154849052429199, "sampling/importance_sampling_ratio/min": 0.11592069268226624, "sampling/importance_sampling_ratio/mean": 1.0226720571517944, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9609599560499191, "clip_ratio/low_mean": 0.08957079518586397, "clip_ratio/low_min": 0.08957079518586397, "clip_ratio/high_mean": 0.08130989596247673, "clip_ratio/high_max": 0.08130989596247673, "clip_ratio/region_mean": 0.1708806911483407, "reward_total_mean": 0.4468875229358673, "reward_meter_mean": 0.29206711053848267, "reward_meter_std": 0.2812570035457611, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9896067380905151, "reward_repeat_soft_std": 0.01452327985316515, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.4468875229358673, "reward_total_composite_std": 0.21935208141803741} {"timestamp_utc": "2026-04-12T22:23:36Z", "mode": "train", "global_step": 69, "epoch": 0.006931190356604721, "loss": 0.1552, "grad_norm": 18.79545783996582, "learning_rate": 9.793939393939394e-06, "num_tokens": 132069.0, "completions/mean_length": 72.375, "completions/min_length": 50.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.375, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.6283866167068481, "rewards/meter/std": 0.29875946044921875, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9875814914703369, "rewards/repeat_soft/std": 0.009496775455772877, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.6875321865081787, "rewards/total_composite/std": 0.15890254080295563, "reward": 0.6875321865081787, "reward_std": 0.15890255570411682, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.26377734541893005, "sampling/sampling_logp_difference/max": 2.8466076850891113, "sampling/importance_sampling_ratio/min": 0.05804087966680527, "sampling/importance_sampling_ratio/mean": 1.0076905488967896, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4112783074378967, "clip_ratio/low_mean": 0.07913436740636826, "clip_ratio/low_min": 0.07913436740636826, "clip_ratio/high_mean": 0.1484571509063244, "clip_ratio/high_max": 0.1484571509063244, "clip_ratio/region_mean": 0.22759151831269264, "reward_total_mean": 0.6875321865081787, "reward_meter_mean": 0.6283866167068481, "reward_meter_std": 0.29875946044921875, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9875814914703369, "reward_repeat_soft_std": 0.009496775455772877, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.6875321865081787, "reward_total_composite_std": 0.15890254080295563} {"timestamp_utc": "2026-04-12T22:23:49Z", "mode": "train", "global_step": 70, "epoch": 0.007031642390758413, "loss": 0.0629, "grad_norm": 11.336478233337402, "learning_rate": 9.790909090909093e-06, "num_tokens": 134316.0, "completions/mean_length": 108.875, "completions/min_length": 50.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.875, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.2835621237754822, "rewards/meter/std": 0.3008175492286682, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9945306777954102, "rewards/repeat_soft/std": 0.003017295151948929, "rewards/judge_quality/mean": 0.6362500190734863, "rewards/judge_quality/std": 0.24047201871871948, "rewards/total_composite/mean": 0.563243567943573, "rewards/total_composite/std": 0.17085498571395874, "reward": 0.563243567943573, "reward_std": 0.17085498571395874, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21989120543003082, "sampling/sampling_logp_difference/max": 1.4525012969970703, "sampling/importance_sampling_ratio/min": 0.23398429155349731, "sampling/importance_sampling_ratio/mean": 1.0469709634780884, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0914076268672943, "clip_ratio/low_mean": 0.18281028605997562, "clip_ratio/low_min": 0.18281028605997562, "clip_ratio/high_mean": 0.05735824815928936, "clip_ratio/high_max": 0.05735824815928936, "clip_ratio/region_mean": 0.24016853421926498, "reward_total_mean": 0.563243567943573, "reward_meter_mean": 0.2835621237754822, "reward_meter_std": 0.3008175492286682, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9945306777954102, "reward_repeat_soft_std": 0.003017295151948929, "reward_judge_quality_mean": 0.6362500190734863, "reward_judge_quality_std": 0.24047201871871948, "reward_total_composite_mean": 0.563243567943573, "reward_total_composite_std": 0.17085498571395874} {"timestamp_utc": "2026-04-12T22:23:58Z", "mode": "train", "global_step": 71, "epoch": 0.007132094424912104, "loss": 0.2256, "grad_norm": 17.786706924438477, "learning_rate": 9.787878787878788e-06, "num_tokens": 136031.0, "completions/mean_length": 42.375, "completions/min_length": 25.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.375, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.6308169364929199, "rewards/meter/std": 0.404389351606369, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9952167272567749, "rewards/repeat_soft/std": 0.004450792912393808, "rewards/judge_quality/mean": 0.7987500429153442, "rewards/judge_quality/std": 0.22465451061725616, "rewards/total_composite/mean": 0.7730143070220947, "rewards/total_composite/std": 0.22425441443920135, "reward": 0.7730143070220947, "reward_std": 0.22425441443920135, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15945900976657867, "sampling/sampling_logp_difference/max": 2.320220947265625, "sampling/importance_sampling_ratio/min": 0.09825187176465988, "sampling/importance_sampling_ratio/mean": 1.0230482816696167, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.073503103107214, "clip_ratio/low_mean": 0.07358044851571321, "clip_ratio/low_min": 0.07358044851571321, "clip_ratio/high_mean": 0.05085150431841612, "clip_ratio/high_max": 0.05085150431841612, "clip_ratio/region_mean": 0.12443195283412933, "reward_total_mean": 0.7730143070220947, "reward_meter_mean": 0.6308169364929199, "reward_meter_std": 0.404389351606369, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9952167272567749, "reward_repeat_soft_std": 0.004450792912393808, "reward_judge_quality_mean": 0.7987500429153442, "reward_judge_quality_std": 0.22465451061725616, "reward_total_composite_mean": 0.7730143070220947, "reward_total_composite_std": 0.22425441443920135} {"timestamp_utc": "2026-04-12T22:24:07Z", "mode": "train", "global_step": 72, "epoch": 0.007232546459065796, "loss": 0.0725, "grad_norm": 12.097882270812988, "learning_rate": 9.784848484848486e-06, "num_tokens": 138022.0, "completions/mean_length": 86.875, "completions/min_length": 77.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.875, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.24359747767448425, "rewards/meter/std": 0.2615220844745636, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.993793249130249, "rewards/repeat_soft/std": 0.002918896032497287, "rewards/judge_quality/mean": 0.6112499833106995, "rewards/judge_quality/std": 0.15037456154823303, "rewards/total_composite/mean": 0.5423731803894043, "rewards/total_composite/std": 0.13441166281700134, "reward": 0.5423731803894043, "reward_std": 0.13441164791584015, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18156975507736206, "sampling/sampling_logp_difference/max": 3.1153430938720703, "sampling/importance_sampling_ratio/min": 0.04436328634619713, "sampling/importance_sampling_ratio/mean": 0.9998382329940796, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9134444296360016, "clip_ratio/low_mean": 0.1097878199070692, "clip_ratio/low_min": 0.1097878199070692, "clip_ratio/high_mean": 0.030120695009827614, "clip_ratio/high_max": 0.030120695009827614, "clip_ratio/region_mean": 0.13990851491689682, "reward_total_mean": 0.5423731803894043, "reward_meter_mean": 0.24359747767448425, "reward_meter_std": 0.2615220844745636, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.993793249130249, "reward_repeat_soft_std": 0.002918896032497287, "reward_judge_quality_mean": 0.6112499833106995, "reward_judge_quality_std": 0.15037456154823303, "reward_total_composite_mean": 0.5423731803894043, "reward_total_composite_std": 0.13441166281700134} {"timestamp_utc": "2026-04-12T22:24:19Z", "mode": "train", "global_step": 73, "epoch": 0.007332998493219488, "loss": 0.0614, "grad_norm": 18.151248931884766, "learning_rate": 9.781818181818183e-06, "num_tokens": 139596.0, "completions/mean_length": 37.75, "completions/min_length": 32.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.75, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.6459850072860718, "rewards/meter/std": 0.36253878474235535, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9962372779846191, "rewards/repeat_soft/std": 0.00649625901132822, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.6873170137405396, "rewards/total_composite/std": 0.13498029112815857, "reward": 0.6873170137405396, "reward_std": 0.13498029112815857, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20438161492347717, "sampling/sampling_logp_difference/max": 1.8445549011230469, "sampling/importance_sampling_ratio/min": 0.15809568762779236, "sampling/importance_sampling_ratio/mean": 1.0211987495422363, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.976985052227974, "clip_ratio/low_mean": 0.07190257450565696, "clip_ratio/low_min": 0.07190257450565696, "clip_ratio/high_mean": 0.08791267033666372, "clip_ratio/high_max": 0.08791267033666372, "clip_ratio/region_mean": 0.15981524484232068, "reward_total_mean": 0.6873170137405396, "reward_meter_mean": 0.6459850072860718, "reward_meter_std": 0.36253878474235535, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9962372779846191, "reward_repeat_soft_std": 0.00649625901132822, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.6873170137405396, "reward_total_composite_std": 0.13498029112815857} {"timestamp_utc": "2026-04-12T22:24:26Z", "mode": "train", "global_step": 74, "epoch": 0.007433450527373179, "loss": 0.0588, "grad_norm": 13.155890464782715, "learning_rate": 9.77878787878788e-06, "num_tokens": 141237.0, "completions/mean_length": 59.125, "completions/min_length": 51.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.125, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.6750097274780273, "rewards/meter/std": 0.36580297350883484, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9978744983673096, "rewards/repeat_soft/std": 0.0031460225582122803, "rewards/judge_quality/mean": 0.7437499761581421, "rewards/judge_quality/std": 0.33683136105537415, "rewards/total_composite/mean": 0.6564773321151733, "rewards/total_composite/std": 0.3408243656158447, "reward": 0.6564773321151733, "reward_std": 0.3408243656158447, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15842394530773163, "sampling/sampling_logp_difference/max": 1.5450174808502197, "sampling/importance_sampling_ratio/min": 0.2133081555366516, "sampling/importance_sampling_ratio/mean": 1.0020114183425903, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8815535008907318, "clip_ratio/low_mean": 0.04728598054498434, "clip_ratio/low_min": 0.04728598054498434, "clip_ratio/high_mean": 0.09232382103800774, "clip_ratio/high_max": 0.09232382103800774, "clip_ratio/region_mean": 0.13960980158299208, "reward_total_mean": 0.6564773321151733, "reward_meter_mean": 0.6750097274780273, "reward_meter_std": 0.36580297350883484, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9978744983673096, "reward_repeat_soft_std": 0.0031460225582122803, "reward_judge_quality_mean": 0.7437499761581421, "reward_judge_quality_std": 0.33683136105537415, "reward_total_composite_mean": 0.6564773321151733, "reward_total_composite_std": 0.3408243656158447} {"timestamp_utc": "2026-04-12T22:24:39Z", "mode": "train", "global_step": 75, "epoch": 0.007533902561526871, "loss": -0.0769, "grad_norm": 9.224102973937988, "learning_rate": 9.775757575757576e-06, "num_tokens": 143471.0, "completions/mean_length": 105.25, "completions/min_length": 74.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.25, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.7617064714431763, "rewards/meter/std": 0.3298294246196747, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9947261214256287, "rewards/repeat_soft/std": 0.006881263107061386, "rewards/judge_quality/mean": 0.5449999570846558, "rewards/judge_quality/std": 0.1752549260854721, "rewards/total_composite/mean": 0.7557405233383179, "rewards/total_composite/std": 0.17117002606391907, "reward": 0.7557405233383179, "reward_std": 0.17117002606391907, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20651639997959137, "sampling/sampling_logp_difference/max": 1.6719465255737305, "sampling/importance_sampling_ratio/min": 0.1878809928894043, "sampling/importance_sampling_ratio/mean": 1.0183568000793457, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3894700407981873, "clip_ratio/low_mean": 0.07153031043708324, "clip_ratio/low_min": 0.07153031043708324, "clip_ratio/high_mean": 0.1301163136959076, "clip_ratio/high_max": 0.1301163136959076, "clip_ratio/region_mean": 0.20164662413299084, "reward_total_mean": 0.7557405233383179, "reward_meter_mean": 0.7617064714431763, "reward_meter_std": 0.3298294246196747, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9947261214256287, "reward_repeat_soft_std": 0.006881263107061386, "reward_judge_quality_mean": 0.5449999570846558, "reward_judge_quality_std": 0.1752549260854721, "reward_total_composite_mean": 0.7557405233383179, "reward_total_composite_std": 0.17117002606391907} {"timestamp_utc": "2026-04-12T22:24:49Z", "mode": "train", "global_step": 76, "epoch": 0.0076343545956805625, "loss": 0.0058, "grad_norm": 18.25312042236328, "learning_rate": 9.772727272727273e-06, "num_tokens": 144964.0, "completions/mean_length": 32.625, "completions/min_length": 25.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.625, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.708471417427063, "rewards/meter/std": 0.4398741126060486, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9486144185066223, "rewards/repeat_soft/std": 0.03010723739862442, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.7481735944747925, "rewards/total_composite/std": 0.20565913617610931, "reward": 0.7481735944747925, "reward_std": 0.20565910637378693, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17192530632019043, "sampling/sampling_logp_difference/max": 0.9567670822143555, "sampling/importance_sampling_ratio/min": 0.3841327428817749, "sampling/importance_sampling_ratio/mean": 1.0192999839782715, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8063572198152542, "clip_ratio/low_mean": 0.06461021397262812, "clip_ratio/low_min": 0.06461021397262812, "clip_ratio/high_mean": 0.10132488049566746, "clip_ratio/high_max": 0.10132488049566746, "clip_ratio/region_mean": 0.16593509446829557, "reward_total_mean": 0.7481735944747925, "reward_meter_mean": 0.708471417427063, "reward_meter_std": 0.4398741126060486, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9486144185066223, "reward_repeat_soft_std": 0.03010723739862442, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.7481735944747925, "reward_total_composite_std": 0.20565913617610931} {"timestamp_utc": "2026-04-12T22:24:57Z", "mode": "train", "global_step": 77, "epoch": 0.0077348066298342545, "loss": -0.0728, "grad_norm": 17.3715763092041, "learning_rate": 9.76969696969697e-06, "num_tokens": 146633.0, "completions/mean_length": 52.625, "completions/min_length": 35.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.625, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.2402770221233368, "rewards/meter/std": 0.31088128685951233, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9913026094436646, "rewards/repeat_soft/std": 0.009482075460255146, "rewards/judge_quality/mean": 0.6862499713897705, "rewards/judge_quality/std": 0.28076112270355225, "rewards/total_composite/mean": 0.5631299018859863, "rewards/total_composite/std": 0.1694323569536209, "reward": 0.5631299018859863, "reward_std": 0.16943234205245972, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1446535289287567, "sampling/sampling_logp_difference/max": 1.915684700012207, "sampling/importance_sampling_ratio/min": 0.14724098145961761, "sampling/importance_sampling_ratio/mean": 1.030705451965332, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0392472222447395, "clip_ratio/low_mean": 0.11356967873871326, "clip_ratio/low_min": 0.11356967873871326, "clip_ratio/high_mean": 0.033929698169231415, "clip_ratio/high_max": 0.033929698169231415, "clip_ratio/region_mean": 0.14749937690794468, "reward_total_mean": 0.5631299018859863, "reward_meter_mean": 0.2402770221233368, "reward_meter_std": 0.31088128685951233, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9913026094436646, "reward_repeat_soft_std": 0.009482075460255146, "reward_judge_quality_mean": 0.6862499713897705, "reward_judge_quality_std": 0.28076112270355225, "reward_total_composite_mean": 0.5631299018859863, "reward_total_composite_std": 0.1694323569536209} {"timestamp_utc": "2026-04-12T22:25:05Z", "mode": "train", "global_step": 78, "epoch": 0.007835258663987946, "loss": 0.0969, "grad_norm": 9.234482765197754, "learning_rate": 9.766666666666667e-06, "num_tokens": 149203.0, "completions/mean_length": 126.25, "completions/min_length": 107.0, "completions/max_length": 153.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.25, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 153.0, "rewards/meter/mean": 0.9073135256767273, "rewards/meter/std": 0.11808735877275467, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9976153373718262, "rewards/repeat_soft/std": 0.0015691010048612952, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.23793382942676544, "rewards/total_composite/mean": 0.8226776123046875, "rewards/total_composite/std": 0.1000823825597763, "reward": 0.8226776123046875, "reward_std": 0.10008236765861511, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19733132421970367, "sampling/sampling_logp_difference/max": 1.7452077865600586, "sampling/importance_sampling_ratio/min": 0.17460869252681732, "sampling/importance_sampling_ratio/mean": 1.0099540948867798, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9953903406858444, "clip_ratio/low_mean": 0.11173281446099281, "clip_ratio/low_min": 0.11173281446099281, "clip_ratio/high_mean": 0.08277627266943455, "clip_ratio/high_max": 0.08277627266943455, "clip_ratio/region_mean": 0.19450908713042736, "reward_total_mean": 0.8226776123046875, "reward_meter_mean": 0.9073135256767273, "reward_meter_std": 0.11808735877275467, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9976153373718262, "reward_repeat_soft_std": 0.0015691010048612952, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.23793382942676544, "reward_total_composite_mean": 0.8226776123046875, "reward_total_composite_std": 0.1000823825597763} {"timestamp_utc": "2026-04-12T22:25:12Z", "mode": "train", "global_step": 79, "epoch": 0.007935710698141637, "loss": -0.0129, "grad_norm": 13.165565490722656, "learning_rate": 9.763636363636365e-06, "num_tokens": 151314.0, "completions/mean_length": 86.875, "completions/min_length": 71.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.875, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.575776219367981, "rewards/meter/std": 0.40430617332458496, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9948314428329468, "rewards/repeat_soft/std": 0.004851632285863161, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.7033324241638184, "rewards/total_composite/std": 0.19417116045951843, "reward": 0.7033324241638184, "reward_std": 0.19417116045951843, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20565244555473328, "sampling/sampling_logp_difference/max": 1.9638829231262207, "sampling/importance_sampling_ratio/min": 0.1403125375509262, "sampling/importance_sampling_ratio/mean": 1.03525710105896, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8156342655420303, "clip_ratio/low_mean": 0.09965229965746403, "clip_ratio/low_min": 0.09965229965746403, "clip_ratio/high_mean": 0.10430697165429592, "clip_ratio/high_max": 0.10430697165429592, "clip_ratio/region_mean": 0.20395927131175995, "reward_total_mean": 0.7033324241638184, "reward_meter_mean": 0.575776219367981, "reward_meter_std": 0.40430617332458496, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9948314428329468, "reward_repeat_soft_std": 0.004851632285863161, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.7033324241638184, "reward_total_composite_std": 0.19417116045951843} {"timestamp_utc": "2026-04-12T22:25:18Z", "mode": "train", "global_step": 80, "epoch": 0.00803616273229533, "loss": -0.017, "grad_norm": 13.996071815490723, "learning_rate": 9.760606060606062e-06, "num_tokens": 153151.0, "completions/mean_length": 60.625, "completions/min_length": 48.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.625, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.786838948726654, "rewards/meter/std": 0.3647293746471405, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9986039400100708, "rewards/repeat_soft/std": 0.0022809384390711784, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.11310552060604095, "rewards/total_composite/mean": 0.7456879615783691, "rewards/total_composite/std": 0.13831521570682526, "reward": 0.7456879615783691, "reward_std": 0.13831521570682526, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21806959807872772, "sampling/sampling_logp_difference/max": 1.4953420162200928, "sampling/importance_sampling_ratio/min": 0.22417192161083221, "sampling/importance_sampling_ratio/mean": 1.045628547668457, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2772906571626663, "clip_ratio/low_mean": 0.05426747165620327, "clip_ratio/low_min": 0.05426747165620327, "clip_ratio/high_mean": 0.14767173863947392, "clip_ratio/high_max": 0.14767173863947392, "clip_ratio/region_mean": 0.20193921029567719, "reward_total_mean": 0.7456879615783691, "reward_meter_mean": 0.786838948726654, "reward_meter_std": 0.3647293746471405, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9986039400100708, "reward_repeat_soft_std": 0.0022809384390711784, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.11310552060604095, "reward_total_composite_mean": 0.7456879615783691, "reward_total_composite_std": 0.13831521570682526} {"timestamp_utc": "2026-04-12T22:25:28Z", "mode": "train", "global_step": 81, "epoch": 0.008136614766449021, "loss": 0.0024, "grad_norm": 10.145329475402832, "learning_rate": 9.757575757575758e-06, "num_tokens": 155821.0, "completions/mean_length": 123.75, "completions/min_length": 77.0, "completions/max_length": 148.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 123.75, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 148.0, "rewards/meter/mean": 0.8844622373580933, "rewards/meter/std": 0.30505692958831787, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9946202039718628, "rewards/repeat_soft/std": 0.009633982554078102, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.7800325155258179, "rewards/total_composite/std": 0.14473092555999756, "reward": 0.7800325155258179, "reward_std": 0.14473092555999756, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20969630777835846, "sampling/sampling_logp_difference/max": 2.1087703704833984, "sampling/importance_sampling_ratio/min": 0.12138713151216507, "sampling/importance_sampling_ratio/mean": 1.0255879163742065, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.578194409608841, "clip_ratio/low_mean": 0.024572649970650673, "clip_ratio/low_min": 0.024572649970650673, "clip_ratio/high_mean": 0.1784006580710411, "clip_ratio/high_max": 0.1784006580710411, "clip_ratio/region_mean": 0.20297330804169178, "reward_total_mean": 0.7800325155258179, "reward_meter_mean": 0.8844622373580933, "reward_meter_std": 0.30505692958831787, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9946202039718628, "reward_repeat_soft_std": 0.009633982554078102, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.7800325155258179, "reward_total_composite_std": 0.14473092555999756} {"timestamp_utc": "2026-04-12T22:25:36Z", "mode": "train", "global_step": 82, "epoch": 0.008237066800602712, "loss": 0.0932, "grad_norm": 15.969931602478027, "learning_rate": 9.754545454545455e-06, "num_tokens": 157487.0, "completions/mean_length": 59.25, "completions/min_length": 49.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.25, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.5042653679847717, "rewards/meter/std": 0.4921252727508545, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9970687627792358, "rewards/repeat_soft/std": 0.004391745664179325, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.6056262850761414, "rewards/total_composite/std": 0.26372030377388, "reward": 0.6056262850761414, "reward_std": 0.26372030377388, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21136270463466644, "sampling/sampling_logp_difference/max": 2.0872573852539062, "sampling/importance_sampling_ratio/min": 0.1240268275141716, "sampling/importance_sampling_ratio/mean": 1.0206705331802368, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.811131864786148, "clip_ratio/low_mean": 0.0915905861184001, "clip_ratio/low_min": 0.0915905861184001, "clip_ratio/high_mean": 0.09937021695077419, "clip_ratio/high_max": 0.09937021695077419, "clip_ratio/region_mean": 0.1909608030691743, "reward_total_mean": 0.6056262850761414, "reward_meter_mean": 0.5042653679847717, "reward_meter_std": 0.4921252727508545, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9970687627792358, "reward_repeat_soft_std": 0.004391745664179325, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.6056262850761414, "reward_total_composite_std": 0.26372030377388} {"timestamp_utc": "2026-04-12T22:25:43Z", "mode": "train", "global_step": 83, "epoch": 0.008337518834756403, "loss": 0.0093, "grad_norm": 13.831247329711914, "learning_rate": 9.751515151515152e-06, "num_tokens": 159104.0, "completions/mean_length": 57.125, "completions/min_length": 40.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.125, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.7920023202896118, "rewards/meter/std": 0.3581661581993103, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9975086450576782, "rewards/repeat_soft/std": 0.003981848247349262, "rewards/judge_quality/mean": 0.5862500071525574, "rewards/judge_quality/std": 0.2822834253311157, "rewards/total_composite/mean": 0.7820268869400024, "rewards/total_composite/std": 0.22069051861763, "reward": 0.7820268869400024, "reward_std": 0.22069051861763, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18509216606616974, "sampling/sampling_logp_difference/max": 1.5339620113372803, "sampling/importance_sampling_ratio/min": 0.21567945182323456, "sampling/importance_sampling_ratio/mean": 1.0375295877456665, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7064232379198074, "clip_ratio/low_mean": 0.044791667722165585, "clip_ratio/low_min": 0.044791667722165585, "clip_ratio/high_mean": 0.13707098085433245, "clip_ratio/high_max": 0.13707098085433245, "clip_ratio/region_mean": 0.18186264857649803, "reward_total_mean": 0.7820268869400024, "reward_meter_mean": 0.7920023202896118, "reward_meter_std": 0.3581661581993103, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9975086450576782, "reward_repeat_soft_std": 0.003981848247349262, "reward_judge_quality_mean": 0.5862500071525574, "reward_judge_quality_std": 0.2822834253311157, "reward_total_composite_mean": 0.7820268869400024, "reward_total_composite_std": 0.22069051861763} {"timestamp_utc": "2026-04-12T22:25:51Z", "mode": "train", "global_step": 84, "epoch": 0.008437970868910096, "loss": -0.0464, "grad_norm": 18.266530990600586, "learning_rate": 9.74848484848485e-06, "num_tokens": 160981.0, "completions/mean_length": 53.625, "completions/min_length": 38.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.625, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.674973726272583, "rewards/meter/std": 0.4194721281528473, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9981741905212402, "rewards/repeat_soft/std": 0.0017913100309669971, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.682930588722229, "rewards/total_composite/std": 0.18727268278598785, "reward": 0.682930588722229, "reward_std": 0.18727268278598785, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2229832410812378, "sampling/sampling_logp_difference/max": 1.926133155822754, "sampling/importance_sampling_ratio/min": 0.1657637655735016, "sampling/importance_sampling_ratio/mean": 1.0519226789474487, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2065766006708145, "clip_ratio/low_mean": 0.07867324538528919, "clip_ratio/low_min": 0.07867324538528919, "clip_ratio/high_mean": 0.11280264891684055, "clip_ratio/high_max": 0.11280264891684055, "clip_ratio/region_mean": 0.19147589430212975, "reward_total_mean": 0.682930588722229, "reward_meter_mean": 0.674973726272583, "reward_meter_std": 0.4194721281528473, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9981741905212402, "reward_repeat_soft_std": 0.0017913100309669971, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.682930588722229, "reward_total_composite_std": 0.18727268278598785} {"timestamp_utc": "2026-04-12T22:25:57Z", "mode": "train", "global_step": 85, "epoch": 0.008538422903063787, "loss": 0.1016, "grad_norm": 26.06890106201172, "learning_rate": 9.745454545454547e-06, "num_tokens": 162397.0, "completions/mean_length": 32.0, "completions/min_length": 15.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.5776013731956482, "rewards/meter/std": 0.4645298421382904, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9671874642372131, "rewards/repeat_soft/std": 0.013258260674774647, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5890144109725952, "rewards/total_composite/std": 0.3000996708869934, "reward": 0.5890144109725952, "reward_std": 0.3000996708869934, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2313683032989502, "sampling/sampling_logp_difference/max": 1.7472820281982422, "sampling/importance_sampling_ratio/min": 0.1742469072341919, "sampling/importance_sampling_ratio/mean": 1.0024120807647705, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7302970439195633, "clip_ratio/low_mean": 0.0708547318354249, "clip_ratio/low_min": 0.0708547318354249, "clip_ratio/high_mean": 0.1231324952095747, "clip_ratio/high_max": 0.1231324952095747, "clip_ratio/region_mean": 0.1939872270449996, "reward_total_mean": 0.5890144109725952, "reward_meter_mean": 0.5776013731956482, "reward_meter_std": 0.4645298421382904, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9671874642372131, "reward_repeat_soft_std": 0.013258260674774647, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5890144109725952, "reward_total_composite_std": 0.3000996708869934} {"timestamp_utc": "2026-04-12T22:26:04Z", "mode": "train", "global_step": 86, "epoch": 0.008638874937217478, "loss": 0.0885, "grad_norm": 19.318851470947266, "learning_rate": 9.742424242424244e-06, "num_tokens": 164496.0, "completions/mean_length": 83.375, "completions/min_length": 70.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.375, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.5039986968040466, "rewards/meter/std": 0.3009590804576874, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9974728226661682, "rewards/repeat_soft/std": 0.0023224493488669395, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.5930466651916504, "rewards/total_composite/std": 0.14554795622825623, "reward": 0.5930466651916504, "reward_std": 0.14554795622825623, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.234708771109581, "sampling/sampling_logp_difference/max": 2.4151086807250977, "sampling/importance_sampling_ratio/min": 0.08935762941837311, "sampling/importance_sampling_ratio/mean": 0.986754834651947, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1506693363189697, "clip_ratio/low_mean": 0.07777538895606995, "clip_ratio/low_min": 0.07777538895606995, "clip_ratio/high_mean": 0.1278723981231451, "clip_ratio/high_max": 0.1278723981231451, "clip_ratio/region_mean": 0.20564778707921505, "reward_total_mean": 0.5930466651916504, "reward_meter_mean": 0.5039986968040466, "reward_meter_std": 0.3009590804576874, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9974728226661682, "reward_repeat_soft_std": 0.0023224493488669395, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.5930466651916504, "reward_total_composite_std": 0.14554795622825623} {"timestamp_utc": "2026-04-12T22:26:11Z", "mode": "train", "global_step": 87, "epoch": 0.00873932697137117, "loss": 0.047, "grad_norm": 32.492706298828125, "learning_rate": 9.739393939393941e-06, "num_tokens": 165854.0, "completions/mean_length": 28.75, "completions/min_length": 25.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.75, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.8606054782867432, "rewards/meter/std": 0.29568642377853394, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9594065546989441, "rewards/repeat_soft/std": 0.008749538101255894, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.8177131414413452, "rewards/total_composite/std": 0.17395693063735962, "reward": 0.8177131414413452, "reward_std": 0.17395691573619843, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1784212738275528, "sampling/sampling_logp_difference/max": 1.7252826690673828, "sampling/importance_sampling_ratio/min": 0.17812269926071167, "sampling/importance_sampling_ratio/mean": 1.010380744934082, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4293856024742126, "clip_ratio/low_mean": 0.07695782743394375, "clip_ratio/low_min": 0.07695782743394375, "clip_ratio/high_mean": 0.11240384820848703, "clip_ratio/high_max": 0.11240384820848703, "clip_ratio/region_mean": 0.18936167564243078, "reward_total_mean": 0.8177131414413452, "reward_meter_mean": 0.8606054782867432, "reward_meter_std": 0.29568642377853394, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9594065546989441, "reward_repeat_soft_std": 0.008749538101255894, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.8177131414413452, "reward_total_composite_std": 0.17395693063735962} {"timestamp_utc": "2026-04-12T22:26:17Z", "mode": "train", "global_step": 88, "epoch": 0.008839779005524863, "loss": 0.0314, "grad_norm": 15.27247142791748, "learning_rate": 9.736363636363637e-06, "num_tokens": 167652.0, "completions/mean_length": 63.75, "completions/min_length": 62.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.75, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.6508787870407104, "rewards/meter/std": 0.44921696186065674, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9988118410110474, "rewards/repeat_soft/std": 0.0022364838514477015, "rewards/judge_quality/mean": 0.5387500524520874, "rewards/judge_quality/std": 0.24485784769058228, "rewards/total_composite/mean": 0.7044016122817993, "rewards/total_composite/std": 0.15757334232330322, "reward": 0.7044016122817993, "reward_std": 0.15757334232330322, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20896635949611664, "sampling/sampling_logp_difference/max": 1.6069178581237793, "sampling/importance_sampling_ratio/min": 0.20050464570522308, "sampling/importance_sampling_ratio/mean": 1.016836166381836, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5759567469358444, "clip_ratio/low_mean": 0.0732837338000536, "clip_ratio/low_min": 0.0732837338000536, "clip_ratio/high_mean": 0.13178288005292416, "clip_ratio/high_max": 0.13178288005292416, "clip_ratio/region_mean": 0.20506661385297775, "reward_total_mean": 0.7044016122817993, "reward_meter_mean": 0.6508787870407104, "reward_meter_std": 0.44921696186065674, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9988118410110474, "reward_repeat_soft_std": 0.0022364838514477015, "reward_judge_quality_mean": 0.5387500524520874, "reward_judge_quality_std": 0.24485784769058228, "reward_total_composite_mean": 0.7044016122817993, "reward_total_composite_std": 0.15757334232330322} {"timestamp_utc": "2026-04-12T22:26:24Z", "mode": "train", "global_step": 89, "epoch": 0.008940231039678554, "loss": 0.0346, "grad_norm": 14.53501033782959, "learning_rate": 9.733333333333334e-06, "num_tokens": 169728.0, "completions/mean_length": 94.5, "completions/min_length": 91.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.5, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.7820674180984497, "rewards/meter/std": 0.24537409842014313, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9976636171340942, "rewards/repeat_soft/std": 0.0012526443460956216, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.765196681022644, "rewards/total_composite/std": 0.10896959155797958, "reward": 0.765196681022644, "reward_std": 0.10896960645914078, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14851930737495422, "sampling/sampling_logp_difference/max": 1.868389368057251, "sampling/importance_sampling_ratio/min": 0.15437209606170654, "sampling/importance_sampling_ratio/mean": 1.010309100151062, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7215096801519394, "clip_ratio/low_mean": 0.03587029594928026, "clip_ratio/low_min": 0.03587029594928026, "clip_ratio/high_mean": 0.09394399169832468, "clip_ratio/high_max": 0.09394399169832468, "clip_ratio/region_mean": 0.12981428764760494, "reward_total_mean": 0.765196681022644, "reward_meter_mean": 0.7820674180984497, "reward_meter_std": 0.24537409842014313, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9976636171340942, "reward_repeat_soft_std": 0.0012526443460956216, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.765196681022644, "reward_total_composite_std": 0.10896959155797958} {"timestamp_utc": "2026-04-12T22:26:31Z", "mode": "train", "global_step": 90, "epoch": 0.009040683073832245, "loss": 0.0817, "grad_norm": 29.59337043762207, "learning_rate": 9.730303030303031e-06, "num_tokens": 171248.0, "completions/mean_length": 33.0, "completions/min_length": 21.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.0, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.418997198343277, "rewards/meter/std": 0.3585245907306671, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9966121912002563, "rewards/repeat_soft/std": 0.00797121599316597, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.1865811049938202, "rewards/total_composite/mean": 0.5975849628448486, "rewards/total_composite/std": 0.19449637830257416, "reward": 0.5975849628448486, "reward_std": 0.19449637830257416, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.25027692317962646, "sampling/sampling_logp_difference/max": 2.5391225814819336, "sampling/importance_sampling_ratio/min": 0.07893563061952591, "sampling/importance_sampling_ratio/mean": 0.9951542019844055, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5199098736047745, "clip_ratio/low_mean": 0.09931824635714293, "clip_ratio/low_min": 0.09931824635714293, "clip_ratio/high_mean": 0.11300505325198174, "clip_ratio/high_max": 0.11300505325198174, "clip_ratio/region_mean": 0.21232329960912466, "reward_total_mean": 0.5975849628448486, "reward_meter_mean": 0.418997198343277, "reward_meter_std": 0.3585245907306671, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9966121912002563, "reward_repeat_soft_std": 0.00797121599316597, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.1865811049938202, "reward_total_composite_mean": 0.5975849628448486, "reward_total_composite_std": 0.19449637830257416} {"timestamp_utc": "2026-04-12T22:26:39Z", "mode": "train", "global_step": 91, "epoch": 0.009141135107985936, "loss": 0.1256, "grad_norm": 10.04101848602295, "learning_rate": 9.727272727272728e-06, "num_tokens": 173561.0, "completions/mean_length": 116.125, "completions/min_length": 95.0, "completions/max_length": 150.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.125, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 150.0, "rewards/meter/mean": 0.693138837814331, "rewards/meter/std": 0.36694416403770447, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9948671460151672, "rewards/repeat_soft/std": 0.005335790105164051, "rewards/judge_quality/mean": 0.4737499952316284, "rewards/judge_quality/std": 0.16291432082653046, "rewards/total_composite/mean": 0.6941492557525635, "rewards/total_composite/std": 0.18267148733139038, "reward": 0.6941492557525635, "reward_std": 0.18267148733139038, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21353013813495636, "sampling/sampling_logp_difference/max": 2.810418128967285, "sampling/importance_sampling_ratio/min": 0.060179825872182846, "sampling/importance_sampling_ratio/mean": 1.0399506092071533, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0453978031873703, "clip_ratio/low_mean": 0.05563131347298622, "clip_ratio/low_min": 0.05563131347298622, "clip_ratio/high_mean": 0.15723679400980473, "clip_ratio/high_max": 0.15723679400980473, "clip_ratio/region_mean": 0.21286810748279095, "reward_total_mean": 0.6941492557525635, "reward_meter_mean": 0.693138837814331, "reward_meter_std": 0.36694416403770447, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9948671460151672, "reward_repeat_soft_std": 0.005335790105164051, "reward_judge_quality_mean": 0.4737499952316284, "reward_judge_quality_std": 0.16291432082653046, "reward_total_composite_mean": 0.6941492557525635, "reward_total_composite_std": 0.18267148733139038} {"timestamp_utc": "2026-04-12T22:26:47Z", "mode": "train", "global_step": 92, "epoch": 0.009241587142139629, "loss": 0.0532, "grad_norm": 14.25716781616211, "learning_rate": 9.724242424242426e-06, "num_tokens": 175372.0, "completions/mean_length": 58.375, "completions/min_length": 41.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.375, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.26952993869781494, "rewards/meter/std": 0.3523905575275421, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9935294389724731, "rewards/repeat_soft/std": 0.008154787123203278, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.500016450881958, "rewards/total_composite/std": 0.15747496485710144, "reward": 0.500016450881958, "reward_std": 0.15747497975826263, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.218789741396904, "sampling/sampling_logp_difference/max": 1.3333110809326172, "sampling/importance_sampling_ratio/min": 0.2674177289009094, "sampling/importance_sampling_ratio/mean": 1.0159591436386108, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6216311752796173, "clip_ratio/low_mean": 0.1285328259691596, "clip_ratio/low_min": 0.1285328259691596, "clip_ratio/high_mean": 0.06342516466975212, "clip_ratio/high_max": 0.06342516466975212, "clip_ratio/region_mean": 0.19195799063891172, "reward_total_mean": 0.500016450881958, "reward_meter_mean": 0.26952993869781494, "reward_meter_std": 0.3523905575275421, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9935294389724731, "reward_repeat_soft_std": 0.008154787123203278, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.500016450881958, "reward_total_composite_std": 0.15747496485710144} {"timestamp_utc": "2026-04-12T22:26:54Z", "mode": "train", "global_step": 93, "epoch": 0.00934203917629332, "loss": 0.0745, "grad_norm": 10.899441719055176, "learning_rate": 9.721212121212123e-06, "num_tokens": 177106.0, "completions/mean_length": 69.75, "completions/min_length": 53.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.75, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9947660565376282, "rewards/meter/std": 0.0041707707569003105, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9986412525177002, "rewards/repeat_soft/std": 0.002481638453900814, "rewards/judge_quality/mean": 0.3725000023841858, "rewards/judge_quality/std": 0.11055056750774384, "rewards/total_composite/mean": 0.8092588186264038, "rewards/total_composite/std": 0.033664677292108536, "reward": 0.8092588186264038, "reward_std": 0.033664681017398834, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17616569995880127, "sampling/sampling_logp_difference/max": 1.2230968475341797, "sampling/importance_sampling_ratio/min": 0.2943173050880432, "sampling/importance_sampling_ratio/mean": 1.0400642156600952, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3353325724601746, "clip_ratio/low_mean": 0.038446055725216866, "clip_ratio/low_min": 0.038446055725216866, "clip_ratio/high_mean": 0.14749629236757755, "clip_ratio/high_max": 0.14749629236757755, "clip_ratio/region_mean": 0.18594234809279442, "reward_total_mean": 0.8092588186264038, "reward_meter_mean": 0.9947660565376282, "reward_meter_std": 0.0041707707569003105, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9986412525177002, "reward_repeat_soft_std": 0.002481638453900814, "reward_judge_quality_mean": 0.3725000023841858, "reward_judge_quality_std": 0.11055056750774384, "reward_total_composite_mean": 0.8092588186264038, "reward_total_composite_std": 0.033664677292108536} {"timestamp_utc": "2026-04-12T22:27:00Z", "mode": "train", "global_step": 94, "epoch": 0.009442491210447011, "loss": 0.0462, "grad_norm": 17.971059799194336, "learning_rate": 9.718181818181818e-06, "num_tokens": 178827.0, "completions/mean_length": 37.125, "completions/min_length": 32.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.125, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.41076311469078064, "rewards/meter/std": 0.3964845538139343, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9856504797935486, "rewards/repeat_soft/std": 0.010154510848224163, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.6366584300994873, "rewards/total_composite/std": 0.15718963742256165, "reward": 0.6366584300994873, "reward_std": 0.15718962252140045, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19394990801811218, "sampling/sampling_logp_difference/max": 1.2942476272583008, "sampling/importance_sampling_ratio/min": 0.2741039991378784, "sampling/importance_sampling_ratio/mean": 1.0088485479354858, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3612593337893486, "clip_ratio/low_mean": 0.08777131512761116, "clip_ratio/low_min": 0.08777131512761116, "clip_ratio/high_mean": 0.07181490398943424, "clip_ratio/high_max": 0.07181490398943424, "clip_ratio/region_mean": 0.1595862191170454, "reward_total_mean": 0.6366584300994873, "reward_meter_mean": 0.41076311469078064, "reward_meter_std": 0.3964845538139343, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9856504797935486, "reward_repeat_soft_std": 0.010154510848224163, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.6366584300994873, "reward_total_composite_std": 0.15718963742256165} {"timestamp_utc": "2026-04-12T22:27:08Z", "mode": "train", "global_step": 95, "epoch": 0.009542943244600702, "loss": 0.0122, "grad_norm": 10.587267875671387, "learning_rate": 9.715151515151516e-06, "num_tokens": 181305.0, "completions/mean_length": 126.75, "completions/min_length": 107.0, "completions/max_length": 151.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.75, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 151.0, "rewards/meter/mean": 0.9200213551521301, "rewards/meter/std": 0.1255701780319214, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9956197738647461, "rewards/repeat_soft/std": 0.003646343480795622, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.21293865144252777, "rewards/total_composite/mean": 0.8480715751647949, "rewards/total_composite/std": 0.08985444903373718, "reward": 0.8480715751647949, "reward_std": 0.08985444158315659, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18390880525112152, "sampling/sampling_logp_difference/max": 1.2477107048034668, "sampling/importance_sampling_ratio/min": 0.2871614396572113, "sampling/importance_sampling_ratio/mean": 1.0285959243774414, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9829654097557068, "clip_ratio/low_mean": 0.08566484786570072, "clip_ratio/low_min": 0.08566484786570072, "clip_ratio/high_mean": 0.09545699134469032, "clip_ratio/high_max": 0.09545699134469032, "clip_ratio/region_mean": 0.18112183921039104, "reward_total_mean": 0.8480715751647949, "reward_meter_mean": 0.9200213551521301, "reward_meter_std": 0.1255701780319214, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9956197738647461, "reward_repeat_soft_std": 0.003646343480795622, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.21293865144252777, "reward_total_composite_mean": 0.8480715751647949, "reward_total_composite_std": 0.08985444903373718} {"timestamp_utc": "2026-04-12T22:27:15Z", "mode": "train", "global_step": 96, "epoch": 0.009643395278754395, "loss": 0.1148, "grad_norm": 14.349672317504883, "learning_rate": 9.712121212121213e-06, "num_tokens": 183026.0, "completions/mean_length": 60.125, "completions/min_length": 41.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.125, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.37904417514801025, "rewards/meter/std": 0.36078208684921265, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9731553792953491, "rewards/repeat_soft/std": 0.04177224636077881, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.5731354355812073, "rewards/total_composite/std": 0.18765978515148163, "reward": 0.5731354355812073, "reward_std": 0.18765978515148163, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20568451285362244, "sampling/sampling_logp_difference/max": 1.8774585723876953, "sampling/importance_sampling_ratio/min": 0.15297840535640717, "sampling/importance_sampling_ratio/mean": 1.0121736526489258, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8408035784959793, "clip_ratio/low_mean": 0.09115871600806713, "clip_ratio/low_min": 0.09115871600806713, "clip_ratio/high_mean": 0.10624979063868523, "clip_ratio/high_max": 0.10624979063868523, "clip_ratio/region_mean": 0.19740850664675236, "reward_total_mean": 0.5731354355812073, "reward_meter_mean": 0.37904417514801025, "reward_meter_std": 0.36078208684921265, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9731553792953491, "reward_repeat_soft_std": 0.04177224636077881, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.5731354355812073, "reward_total_composite_std": 0.18765978515148163} {"timestamp_utc": "2026-04-12T22:27:21Z", "mode": "train", "global_step": 97, "epoch": 0.009743847312908087, "loss": -0.0658, "grad_norm": 13.329829216003418, "learning_rate": 9.70909090909091e-06, "num_tokens": 184785.0, "completions/mean_length": 54.875, "completions/min_length": 35.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.875, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.44293856620788574, "rewards/meter/std": 0.3831910789012909, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9962660074234009, "rewards/repeat_soft/std": 0.004540997091680765, "rewards/judge_quality/mean": 0.38874998688697815, "rewards/judge_quality/std": 0.08675704896450043, "rewards/total_composite/mean": 0.4797763526439667, "rewards/total_composite/std": 0.24937774240970612, "reward": 0.4797763526439667, "reward_std": 0.24937772750854492, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18477389216423035, "sampling/sampling_logp_difference/max": 1.34785795211792, "sampling/importance_sampling_ratio/min": 0.2597961723804474, "sampling/importance_sampling_ratio/mean": 1.0322825908660889, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9412241727113724, "clip_ratio/low_mean": 0.07615137845277786, "clip_ratio/low_min": 0.07615137845277786, "clip_ratio/high_mean": 0.08887524623423815, "clip_ratio/high_max": 0.08887524623423815, "clip_ratio/region_mean": 0.165026624687016, "reward_total_mean": 0.4797763526439667, "reward_meter_mean": 0.44293856620788574, "reward_meter_std": 0.3831910789012909, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9962660074234009, "reward_repeat_soft_std": 0.004540997091680765, "reward_judge_quality_mean": 0.38874998688697815, "reward_judge_quality_std": 0.08675704896450043, "reward_total_composite_mean": 0.4797763526439667, "reward_total_composite_std": 0.24937774240970612} {"timestamp_utc": "2026-04-12T22:27:28Z", "mode": "train", "global_step": 98, "epoch": 0.009844299347061778, "loss": 0.0294, "grad_norm": 10.771571159362793, "learning_rate": 9.706060606060606e-06, "num_tokens": 186626.0, "completions/mean_length": 66.125, "completions/min_length": 57.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.125, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.6990766525268555, "rewards/meter/std": 0.38784509897232056, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9952908754348755, "rewards/repeat_soft/std": 0.007525984197854996, "rewards/judge_quality/mean": 0.4762499928474426, "rewards/judge_quality/std": 0.19167961180210114, "rewards/total_composite/mean": 0.7069886326789856, "rewards/total_composite/std": 0.19634383916854858, "reward": 0.7069886326789856, "reward_std": 0.19634383916854858, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19846424460411072, "sampling/sampling_logp_difference/max": 1.7335278987884521, "sampling/importance_sampling_ratio/min": 0.17666007578372955, "sampling/importance_sampling_ratio/mean": 1.0128724575042725, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7142320424318314, "clip_ratio/low_mean": 0.04686454962939024, "clip_ratio/low_min": 0.04686454962939024, "clip_ratio/high_mean": 0.12352950684726238, "clip_ratio/high_max": 0.12352950684726238, "clip_ratio/region_mean": 0.17039405647665262, "reward_total_mean": 0.7069886326789856, "reward_meter_mean": 0.6990766525268555, "reward_meter_std": 0.38784509897232056, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9952908754348755, "reward_repeat_soft_std": 0.007525984197854996, "reward_judge_quality_mean": 0.4762499928474426, "reward_judge_quality_std": 0.19167961180210114, "reward_total_composite_mean": 0.7069886326789856, "reward_total_composite_std": 0.19634383916854858} {"timestamp_utc": "2026-04-12T22:27:35Z", "mode": "train", "global_step": 99, "epoch": 0.009944751381215469, "loss": 0.0833, "grad_norm": 17.894399642944336, "learning_rate": 9.703030303030305e-06, "num_tokens": 188002.0, "completions/mean_length": 28.0, "completions/min_length": 21.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.0, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.49489158391952515, "rewards/meter/std": 0.4102099537849426, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9463370442390442, "rewards/repeat_soft/std": 0.02908223308622837, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.5978349447250366, "rewards/total_composite/std": 0.18156224489212036, "reward": 0.5978349447250366, "reward_std": 0.18156225979328156, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.172475203871727, "sampling/sampling_logp_difference/max": 1.023299217224121, "sampling/importance_sampling_ratio/min": 0.3594072461128235, "sampling/importance_sampling_ratio/mean": 1.049785852432251, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2047024965286255, "clip_ratio/low_mean": 0.08318580687046051, "clip_ratio/low_min": 0.08318580687046051, "clip_ratio/high_mean": 0.07235023286193609, "clip_ratio/high_max": 0.07235023286193609, "clip_ratio/region_mean": 0.1555360397323966, "reward_total_mean": 0.5978349447250366, "reward_meter_mean": 0.49489158391952515, "reward_meter_std": 0.4102099537849426, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9463370442390442, "reward_repeat_soft_std": 0.02908223308622837, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.5978349447250366, "reward_total_composite_std": 0.18156224489212036} {"timestamp_utc": "2026-04-12T22:27:42Z", "mode": "train", "global_step": 100, "epoch": 0.010045203415369162, "loss": 0.0883, "grad_norm": 10.852298736572266, "learning_rate": 9.7e-06, "num_tokens": 190092.0, "completions/mean_length": 87.25, "completions/min_length": 70.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.25, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.25189706683158875, "rewards/meter/std": 0.2712648808956146, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9939076900482178, "rewards/repeat_soft/std": 0.004518185276538134, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.5112444162368774, "rewards/total_composite/std": 0.12693306803703308, "reward": 0.5112444162368774, "reward_std": 0.1269330531358719, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21233618259429932, "sampling/sampling_logp_difference/max": 1.5831584930419922, "sampling/importance_sampling_ratio/min": 0.20532555878162384, "sampling/importance_sampling_ratio/mean": 1.0322673320770264, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8813354671001434, "clip_ratio/low_mean": 0.14453337900340557, "clip_ratio/low_min": 0.14453337900340557, "clip_ratio/high_mean": 0.053007518872618675, "clip_ratio/high_max": 0.053007518872618675, "clip_ratio/region_mean": 0.19754089787602425, "reward_total_mean": 0.5112444162368774, "reward_meter_mean": 0.25189706683158875, "reward_meter_std": 0.2712648808956146, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9939076900482178, "reward_repeat_soft_std": 0.004518185276538134, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.5112444162368774, "reward_total_composite_std": 0.12693306803703308} {"timestamp_utc": "2026-04-12T22:28:35Z", "mode": "eval", "global_step": 100, "epoch": 0.010045203415369162, "eval_loss": NaN, "eval_runtime": 53.4733, "eval_samples_per_second": 1.496, "eval_steps_per_second": 0.187, "eval_num_tokens": 190092.0, "eval_completions/mean_length": 89.825, "eval_completions/min_length": 35.7, "eval_completions/max_length": 160.4, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 89.825, "eval_completions/min_terminated_length": 35.7, "eval_completions/max_terminated_length": 160.4, "eval_rewards/meter/mean": 0.5703477174043655, "eval_rewards/meter/std": 0.38051448166370394, "eval_rewards/count_adherence/mean": 0.9891666650772095, "eval_rewards/count_adherence/std": 0.025364020839333534, "eval_rewards/hard_gate/mean": 0.9125, "eval_rewards/hard_gate/std": 0.2230676978826523, "eval_rewards/repeat_soft/mean": 0.9923222184181213, "eval_rewards/repeat_soft/std": 0.011972753837471827, "eval_rewards/judge_quality/mean": 0.48449999690055845, "eval_rewards/judge_quality/std": 0.16168474704027175, "eval_rewards/total_composite/mean": 0.595814323425293, "eval_rewards/total_composite/std": 0.2441583454608917, "eval_reward": 0.595814323425293, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.13042871057987213, "eval_sampling/sampling_logp_difference/max": 1.1281907081604003, "eval_sampling/importance_sampling_ratio/min": 0.3298598065972328, "eval_sampling/importance_sampling_ratio/mean": 1.0367148160934447, "eval_sampling/importance_sampling_ratio/max": 1.5091773986816406, "eval_entropy": 1.9399710893630981, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.595814323425293, "eval_reward_meter_mean": 0.5703477174043655, "eval_reward_meter_std": 0.38051448166370394, "eval_reward_count_adherence_mean": 0.9891666650772095, "eval_reward_count_adherence_std": 0.025364020839333534, "eval_reward_hard_gate_mean": 0.9125, "eval_reward_hard_gate_std": 0.2230676978826523, "eval_reward_repeat_soft_mean": 0.9923222184181213, "eval_reward_repeat_soft_std": 0.011972753837471827, "eval_reward_judge_quality_mean": 0.48449999690055845, "eval_reward_judge_quality_std": 0.16168474704027175, "eval_reward_total_composite_mean": 0.595814323425293, "eval_reward_total_composite_std": 0.2441583454608917} {"timestamp_utc": "2026-04-12T22:28:46Z", "mode": "train", "global_step": 101, "epoch": 0.010145655449522853, "loss": 0.0307, "grad_norm": 8.720218658447266, "learning_rate": 9.696969696969698e-06, "num_tokens": 192271.0, "completions/mean_length": 96.375, "completions/min_length": 83.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.375, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.9825379252433777, "rewards/meter/std": 0.03238074481487274, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9823436737060547, "rewards/repeat_soft/std": 0.021364767104387283, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.8100014328956604, "rewards/total_composite/std": 0.02203506976366043, "reward": 0.8100014328956604, "reward_std": 0.02203506790101528, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19199949502944946, "sampling/sampling_logp_difference/max": 1.4776219129562378, "sampling/importance_sampling_ratio/min": 0.2281796783208847, "sampling/importance_sampling_ratio/mean": 1.0370380878448486, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.443674087524414, "clip_ratio/low_mean": 0.04417553171515465, "clip_ratio/low_min": 0.04417553171515465, "clip_ratio/high_mean": 0.1387271974235773, "clip_ratio/high_max": 0.1387271974235773, "clip_ratio/region_mean": 0.18290272913873196, "reward_total_mean": 0.8100014328956604, "reward_meter_mean": 0.9825379252433777, "reward_meter_std": 0.03238074481487274, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9823436737060547, "reward_repeat_soft_std": 0.021364767104387283, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.8100014328956604, "reward_total_composite_std": 0.02203506976366043} {"timestamp_utc": "2026-04-12T22:28:58Z", "mode": "train", "global_step": 102, "epoch": 0.010246107483676544, "loss": 0.0103, "grad_norm": 10.914584159851074, "learning_rate": 9.693939393939395e-06, "num_tokens": 194566.0, "completions/mean_length": 107.875, "completions/min_length": 84.0, "completions/max_length": 141.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.875, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 141.0, "rewards/meter/mean": 0.4683343768119812, "rewards/meter/std": 0.3329935371875763, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9961557388305664, "rewards/repeat_soft/std": 0.0035384881775826216, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.5976160168647766, "rewards/total_composite/std": 0.17027634382247925, "reward": 0.5976160168647766, "reward_std": 0.17027634382247925, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22297446429729462, "sampling/sampling_logp_difference/max": 2.2250664234161377, "sampling/importance_sampling_ratio/min": 0.10806024074554443, "sampling/importance_sampling_ratio/mean": 1.0301787853240967, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9477074295282364, "clip_ratio/low_mean": 0.14782720245420933, "clip_ratio/low_min": 0.14782720245420933, "clip_ratio/high_mean": 0.08563361689448357, "clip_ratio/high_max": 0.08563361689448357, "clip_ratio/region_mean": 0.2334608193486929, "reward_total_mean": 0.5976160168647766, "reward_meter_mean": 0.4683343768119812, "reward_meter_std": 0.3329935371875763, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9961557388305664, "reward_repeat_soft_std": 0.0035384881775826216, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.5976160168647766, "reward_total_composite_std": 0.17027634382247925} {"timestamp_utc": "2026-04-12T22:29:08Z", "mode": "train", "global_step": 103, "epoch": 0.010346559517830235, "loss": 0.0549, "grad_norm": 17.430255889892578, "learning_rate": 9.690909090909092e-06, "num_tokens": 196322.0, "completions/mean_length": 62.5, "completions/min_length": 59.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.5, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.8165329694747925, "rewards/meter/std": 0.3091726303100586, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9988213777542114, "rewards/repeat_soft/std": 0.0023674487601965666, "rewards/judge_quality/mean": 0.6187499761581421, "rewards/judge_quality/std": 0.24976776540279388, "rewards/total_composite/mean": 0.802946925163269, "rewards/total_composite/std": 0.1676766276359558, "reward": 0.802946925163269, "reward_std": 0.1676766276359558, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1588955819606781, "sampling/sampling_logp_difference/max": 1.617774486541748, "sampling/importance_sampling_ratio/min": 0.19833961129188538, "sampling/importance_sampling_ratio/mean": 1.0459027290344238, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1695427224040031, "clip_ratio/low_mean": 0.02010148297995329, "clip_ratio/low_min": 0.02010148297995329, "clip_ratio/high_mean": 0.09783541085198522, "clip_ratio/high_max": 0.09783541085198522, "clip_ratio/region_mean": 0.1179368938319385, "reward_total_mean": 0.802946925163269, "reward_meter_mean": 0.8165329694747925, "reward_meter_std": 0.3091726303100586, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9988213777542114, "reward_repeat_soft_std": 0.0023674487601965666, "reward_judge_quality_mean": 0.6187499761581421, "reward_judge_quality_std": 0.24976776540279388, "reward_total_composite_mean": 0.802946925163269, "reward_total_composite_std": 0.1676766276359558} {"timestamp_utc": "2026-04-12T22:29:14Z", "mode": "train", "global_step": 104, "epoch": 0.010447011551983928, "loss": 0.0394, "grad_norm": 15.631914138793945, "learning_rate": 9.687878787878788e-06, "num_tokens": 198005.0, "completions/mean_length": 56.375, "completions/min_length": 50.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.375, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9202769994735718, "rewards/meter/std": 0.12859795987606049, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9968171715736389, "rewards/repeat_soft/std": 0.00538486847653985, "rewards/judge_quality/mean": 0.5600000023841858, "rewards/judge_quality/std": 0.22258226573467255, "rewards/total_composite/mean": 0.8318063616752625, "rewards/total_composite/std": 0.05040104314684868, "reward": 0.8318063616752625, "reward_std": 0.05040103569626808, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18321827054023743, "sampling/sampling_logp_difference/max": 1.7233179807662964, "sampling/importance_sampling_ratio/min": 0.17847299575805664, "sampling/importance_sampling_ratio/mean": 1.009376049041748, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4439412578940392, "clip_ratio/low_mean": 0.13680845219641924, "clip_ratio/low_min": 0.13680845219641924, "clip_ratio/high_mean": 0.04409965127706528, "clip_ratio/high_max": 0.04409965127706528, "clip_ratio/region_mean": 0.18090810347348452, "reward_total_mean": 0.8318063616752625, "reward_meter_mean": 0.9202769994735718, "reward_meter_std": 0.12859795987606049, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9968171715736389, "reward_repeat_soft_std": 0.00538486847653985, "reward_judge_quality_mean": 0.5600000023841858, "reward_judge_quality_std": 0.22258226573467255, "reward_total_composite_mean": 0.8318063616752625, "reward_total_composite_std": 0.05040104314684868} {"timestamp_utc": "2026-04-12T22:29:21Z", "mode": "train", "global_step": 105, "epoch": 0.01054746358613762, "loss": -0.0037, "grad_norm": 8.602232933044434, "learning_rate": 9.684848484848487e-06, "num_tokens": 200456.0, "completions/mean_length": 122.375, "completions/min_length": 105.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.375, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.618858814239502, "rewards/meter/std": 0.37144485116004944, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9973665475845337, "rewards/repeat_soft/std": 0.0020187776535749435, "rewards/judge_quality/mean": 0.39375001192092896, "rewards/judge_quality/std": 0.15638209879398346, "rewards/total_composite/mean": 0.5785918235778809, "rewards/total_composite/std": 0.2889094352722168, "reward": 0.5785918235778809, "reward_std": 0.2889094650745392, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18066643178462982, "sampling/sampling_logp_difference/max": 1.9373149871826172, "sampling/importance_sampling_ratio/min": 0.14409030973911285, "sampling/importance_sampling_ratio/mean": 1.02179753780365, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8448036909103394, "clip_ratio/low_mean": 0.08731714449822903, "clip_ratio/low_min": 0.08731714449822903, "clip_ratio/high_mean": 0.13265781849622726, "clip_ratio/high_max": 0.13265781849622726, "clip_ratio/region_mean": 0.2199749629944563, "reward_total_mean": 0.5785918235778809, "reward_meter_mean": 0.618858814239502, "reward_meter_std": 0.37144485116004944, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9973665475845337, "reward_repeat_soft_std": 0.0020187776535749435, "reward_judge_quality_mean": 0.39375001192092896, "reward_judge_quality_std": 0.15638209879398346, "reward_total_composite_mean": 0.5785918235778809, "reward_total_composite_std": 0.2889094352722168} {"timestamp_utc": "2026-04-12T22:29:29Z", "mode": "train", "global_step": 106, "epoch": 0.01064791562029131, "loss": 0.1191, "grad_norm": 20.697811126708984, "learning_rate": 9.681818181818182e-06, "num_tokens": 201933.0, "completions/mean_length": 32.625, "completions/min_length": 27.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.625, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.6116139888763428, "rewards/meter/std": 0.38843685388565063, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9910503625869751, "rewards/repeat_soft/std": 0.013928886502981186, "rewards/judge_quality/mean": 0.5149999856948853, "rewards/judge_quality/std": 0.2676885426044464, "rewards/total_composite/mean": 0.5906954407691956, "rewards/total_composite/std": 0.3208555281162262, "reward": 0.5906954407691956, "reward_std": 0.3208554983139038, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20241546630859375, "sampling/sampling_logp_difference/max": 1.8645049333572388, "sampling/importance_sampling_ratio/min": 0.15497291088104248, "sampling/importance_sampling_ratio/mean": 0.982658863067627, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9796655997633934, "clip_ratio/low_mean": 0.07287175673991442, "clip_ratio/low_min": 0.07287175673991442, "clip_ratio/high_mean": 0.08723545260727406, "clip_ratio/high_max": 0.08723545260727406, "clip_ratio/region_mean": 0.16010720934718847, "reward_total_mean": 0.5906954407691956, "reward_meter_mean": 0.6116139888763428, "reward_meter_std": 0.38843685388565063, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9910503625869751, "reward_repeat_soft_std": 0.013928886502981186, "reward_judge_quality_mean": 0.5149999856948853, "reward_judge_quality_std": 0.2676885426044464, "reward_total_composite_mean": 0.5906954407691956, "reward_total_composite_std": 0.3208555281162262} {"timestamp_utc": "2026-04-12T22:29:35Z", "mode": "train", "global_step": 107, "epoch": 0.010748367654445002, "loss": 0.0329, "grad_norm": 10.300890922546387, "learning_rate": 9.67878787878788e-06, "num_tokens": 203724.0, "completions/mean_length": 60.875, "completions/min_length": 55.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.6681768894195557, "rewards/meter/std": 0.33801743388175964, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9916098713874817, "rewards/repeat_soft/std": 0.010880302637815475, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.2499571591615677, "rewards/total_composite/mean": 0.7080905437469482, "rewards/total_composite/std": 0.16318832337856293, "reward": 0.7080905437469482, "reward_std": 0.16318833827972412, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18917357921600342, "sampling/sampling_logp_difference/max": 1.667069435119629, "sampling/importance_sampling_ratio/min": 0.18879954516887665, "sampling/importance_sampling_ratio/mean": 0.9972324371337891, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1798932030797005, "clip_ratio/low_mean": 0.0705209244042635, "clip_ratio/low_min": 0.0705209244042635, "clip_ratio/high_mean": 0.0801126342266798, "clip_ratio/high_max": 0.0801126342266798, "clip_ratio/region_mean": 0.1506335586309433, "reward_total_mean": 0.7080905437469482, "reward_meter_mean": 0.6681768894195557, "reward_meter_std": 0.33801743388175964, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9916098713874817, "reward_repeat_soft_std": 0.010880302637815475, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.2499571591615677, "reward_total_composite_mean": 0.7080905437469482, "reward_total_composite_std": 0.16318832337856293} {"timestamp_utc": "2026-04-12T22:29:44Z", "mode": "train", "global_step": 108, "epoch": 0.010848819688598695, "loss": -0.0065, "grad_norm": 25.379682540893555, "learning_rate": 9.675757575757577e-06, "num_tokens": 205061.0, "completions/mean_length": 32.125, "completions/min_length": 23.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.125, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.28980332612991333, "rewards/meter/std": 0.2669162452220917, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9977822303771973, "rewards/repeat_soft/std": 0.0052426340989768505, "rewards/judge_quality/mean": 0.7400000095367432, "rewards/judge_quality/std": 0.24859607219696045, "rewards/total_composite/mean": 0.6021897196769714, "rewards/total_composite/std": 0.12454845756292343, "reward": 0.6021897196769714, "reward_std": 0.12454845756292343, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24968045949935913, "sampling/sampling_logp_difference/max": 2.7263078689575195, "sampling/importance_sampling_ratio/min": 0.06546053290367126, "sampling/importance_sampling_ratio/mean": 1.0117802619934082, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7245535105466843, "clip_ratio/low_mean": 0.1483722161501646, "clip_ratio/low_min": 0.1483722161501646, "clip_ratio/high_mean": 0.06614910159260035, "clip_ratio/high_max": 0.06614910159260035, "clip_ratio/region_mean": 0.21452131774276495, "reward_total_mean": 0.6021897196769714, "reward_meter_mean": 0.28980332612991333, "reward_meter_std": 0.2669162452220917, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9977822303771973, "reward_repeat_soft_std": 0.0052426340989768505, "reward_judge_quality_mean": 0.7400000095367432, "reward_judge_quality_std": 0.24859607219696045, "reward_total_composite_mean": 0.6021897196769714, "reward_total_composite_std": 0.12454845756292343} {"timestamp_utc": "2026-04-12T22:29:51Z", "mode": "train", "global_step": 109, "epoch": 0.010949271722752386, "loss": -0.0321, "grad_norm": 17.32401466369629, "learning_rate": 9.672727272727274e-06, "num_tokens": 206731.0, "completions/mean_length": 48.75, "completions/min_length": 35.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.75, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.8012364506721497, "rewards/meter/std": 0.29985368251800537, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9873402118682861, "rewards/repeat_soft/std": 0.007849778980016708, "rewards/judge_quality/mean": 0.5487499833106995, "rewards/judge_quality/std": 0.22937417030334473, "rewards/total_composite/mean": 0.6714390516281128, "rewards/total_composite/std": 0.3215502202510834, "reward": 0.6714390516281128, "reward_std": 0.3215502202510834, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21072760224342346, "sampling/sampling_logp_difference/max": 1.7081098556518555, "sampling/importance_sampling_ratio/min": 0.18120796978473663, "sampling/importance_sampling_ratio/mean": 1.0020984411239624, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.047327294945717, "clip_ratio/low_mean": 0.062176925130188465, "clip_ratio/low_min": 0.062176925130188465, "clip_ratio/high_mean": 0.15053118951618671, "clip_ratio/high_max": 0.15053118951618671, "clip_ratio/region_mean": 0.21270811464637518, "reward_total_mean": 0.6714390516281128, "reward_meter_mean": 0.8012364506721497, "reward_meter_std": 0.29985368251800537, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9873402118682861, "reward_repeat_soft_std": 0.007849778980016708, "reward_judge_quality_mean": 0.5487499833106995, "reward_judge_quality_std": 0.22937417030334473, "reward_total_composite_mean": 0.6714390516281128, "reward_total_composite_std": 0.3215502202510834} {"timestamp_utc": "2026-04-12T22:29:58Z", "mode": "train", "global_step": 110, "epoch": 0.011049723756906077, "loss": 0.0301, "grad_norm": 12.794631004333496, "learning_rate": 9.66969696969697e-06, "num_tokens": 208570.0, "completions/mean_length": 60.875, "completions/min_length": 35.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.881321907043457, "rewards/meter/std": 0.29402533173561096, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9829816818237305, "rewards/repeat_soft/std": 0.019399171695113182, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.7753930687904358, "rewards/total_composite/std": 0.13116681575775146, "reward": 0.7753930687904358, "reward_std": 0.13116681575775146, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1848183274269104, "sampling/sampling_logp_difference/max": 1.5729122161865234, "sampling/importance_sampling_ratio/min": 0.20744018256664276, "sampling/importance_sampling_ratio/mean": 1.0221983194351196, "sampling/importance_sampling_ratio/max": 1.9403345584869385, "entropy": 1.8935251384973526, "clip_ratio/low_mean": 0.02330508455634117, "clip_ratio/low_min": 0.02330508455634117, "clip_ratio/high_mean": 0.1399045344442129, "clip_ratio/high_max": 0.1399045344442129, "clip_ratio/region_mean": 0.16320961900055408, "reward_total_mean": 0.7753930687904358, "reward_meter_mean": 0.881321907043457, "reward_meter_std": 0.29402533173561096, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9829816818237305, "reward_repeat_soft_std": 0.019399171695113182, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.7753930687904358, "reward_total_composite_std": 0.13116681575775146} {"timestamp_utc": "2026-04-12T22:30:06Z", "mode": "train", "global_step": 111, "epoch": 0.01115017579105977, "loss": -0.049, "grad_norm": 8.582052230834961, "learning_rate": 9.666666666666667e-06, "num_tokens": 210836.0, "completions/mean_length": 119.25, "completions/min_length": 87.0, "completions/max_length": 137.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.25, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.6979060173034668, "rewards/meter/std": 0.3188508450984955, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9957622289657593, "rewards/repeat_soft/std": 0.005219956859946251, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.1524970829486847, "rewards/total_composite/mean": 0.6907588839530945, "rewards/total_composite/std": 0.15496724843978882, "reward": 0.6907588839530945, "reward_std": 0.15496723353862762, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19657625257968903, "sampling/sampling_logp_difference/max": 1.2601728439331055, "sampling/importance_sampling_ratio/min": 0.28360500931739807, "sampling/importance_sampling_ratio/mean": 1.035998821258545, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.517204165458679, "clip_ratio/low_mean": 0.09810720011591911, "clip_ratio/low_min": 0.09810720011591911, "clip_ratio/high_mean": 0.09997841343283653, "clip_ratio/high_max": 0.09997841343283653, "clip_ratio/region_mean": 0.19808561354875565, "reward_total_mean": 0.6907588839530945, "reward_meter_mean": 0.6979060173034668, "reward_meter_std": 0.3188508450984955, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9957622289657593, "reward_repeat_soft_std": 0.005219956859946251, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.1524970829486847, "reward_total_composite_mean": 0.6907588839530945, "reward_total_composite_std": 0.15496724843978882} {"timestamp_utc": "2026-04-12T22:30:12Z", "mode": "train", "global_step": 112, "epoch": 0.011250627825213461, "loss": 0.0122, "grad_norm": 20.0980224609375, "learning_rate": 9.663636363636364e-06, "num_tokens": 212352.0, "completions/mean_length": 24.5, "completions/min_length": 17.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.5, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.5157471299171448, "rewards/meter/std": 0.40246090292930603, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.5137500166893005, "rewards/judge_quality/std": 0.26462578773498535, "rewards/total_composite/mean": 0.6324611902236938, "rewards/total_composite/std": 0.1423051655292511, "reward": 0.6324611902236938, "reward_std": 0.1423051506280899, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1752002090215683, "sampling/sampling_logp_difference/max": 1.0431177616119385, "sampling/importance_sampling_ratio/min": 0.35235440731048584, "sampling/importance_sampling_ratio/mean": 1.0118403434753418, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3280086815357208, "clip_ratio/low_mean": 0.09726544190198183, "clip_ratio/low_min": 0.09726544190198183, "clip_ratio/high_mean": 0.05161340953782201, "clip_ratio/high_max": 0.05161340953782201, "clip_ratio/region_mean": 0.14887885143980384, "reward_total_mean": 0.6324611902236938, "reward_meter_mean": 0.5157471299171448, "reward_meter_std": 0.40246090292930603, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.5137500166893005, "reward_judge_quality_std": 0.26462578773498535, "reward_total_composite_mean": 0.6324611902236938, "reward_total_composite_std": 0.1423051655292511} {"timestamp_utc": "2026-04-12T22:30:18Z", "mode": "train", "global_step": 113, "epoch": 0.011351079859367152, "loss": 0.0815, "grad_norm": 12.603521347045898, "learning_rate": 9.660606060606061e-06, "num_tokens": 214242.0, "completions/mean_length": 80.25, "completions/min_length": 72.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.25, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.35283008217811584, "rewards/meter/std": 0.32658228278160095, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9970334768295288, "rewards/repeat_soft/std": 0.004829896613955498, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.263435423374176, "rewards/total_composite/mean": 0.6106019020080566, "rewards/total_composite/std": 0.17289768159389496, "reward": 0.6106019020080566, "reward_std": 0.17289769649505615, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18053029477596283, "sampling/sampling_logp_difference/max": 2.1333065032958984, "sampling/importance_sampling_ratio/min": 0.1184450089931488, "sampling/importance_sampling_ratio/mean": 1.0171382427215576, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4742966517806053, "clip_ratio/low_mean": 0.07181645464152098, "clip_ratio/low_min": 0.07181645464152098, "clip_ratio/high_mean": 0.08113026432693005, "clip_ratio/high_max": 0.08113026432693005, "clip_ratio/region_mean": 0.15294671896845102, "reward_total_mean": 0.6106019020080566, "reward_meter_mean": 0.35283008217811584, "reward_meter_std": 0.32658228278160095, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9970334768295288, "reward_repeat_soft_std": 0.004829896613955498, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.263435423374176, "reward_total_composite_mean": 0.6106019020080566, "reward_total_composite_std": 0.17289768159389496} {"timestamp_utc": "2026-04-12T22:30:25Z", "mode": "train", "global_step": 114, "epoch": 0.011451531893520843, "loss": 0.0679, "grad_norm": 13.874102592468262, "learning_rate": 9.657575757575758e-06, "num_tokens": 215968.0, "completions/mean_length": 61.75, "completions/min_length": 44.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.75, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.555526614189148, "rewards/meter/std": 0.47907620668411255, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.994547963142395, "rewards/repeat_soft/std": 0.013426722027361393, "rewards/judge_quality/mean": 0.35999998450279236, "rewards/judge_quality/std": 0.09165150672197342, "rewards/total_composite/mean": 0.607441782951355, "rewards/total_composite/std": 0.2246592491865158, "reward": 0.607441782951355, "reward_std": 0.2246592491865158, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19817504286766052, "sampling/sampling_logp_difference/max": 1.2710886001586914, "sampling/importance_sampling_ratio/min": 0.2805260717868805, "sampling/importance_sampling_ratio/mean": 1.0383985042572021, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0831169039011, "clip_ratio/low_mean": 0.07543539721518755, "clip_ratio/low_min": 0.07543539721518755, "clip_ratio/high_mean": 0.10498625971376896, "clip_ratio/high_max": 0.10498625971376896, "clip_ratio/region_mean": 0.1804216569289565, "reward_total_mean": 0.607441782951355, "reward_meter_mean": 0.555526614189148, "reward_meter_std": 0.47907620668411255, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.994547963142395, "reward_repeat_soft_std": 0.013426722027361393, "reward_judge_quality_mean": 0.35999998450279236, "reward_judge_quality_std": 0.09165150672197342, "reward_total_composite_mean": 0.607441782951355, "reward_total_composite_std": 0.2246592491865158} {"timestamp_utc": "2026-04-12T22:30:33Z", "mode": "train", "global_step": 115, "epoch": 0.011551983927674536, "loss": -0.012, "grad_norm": 9.302298545837402, "learning_rate": 9.654545454545456e-06, "num_tokens": 218404.0, "completions/mean_length": 116.5, "completions/min_length": 95.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.5, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.5344254970550537, "rewards/meter/std": 0.3820936977863312, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.99835604429245, "rewards/repeat_soft/std": 0.0023890030570328236, "rewards/judge_quality/mean": 0.47749996185302734, "rewards/judge_quality/std": 0.16263456642627716, "rewards/total_composite/mean": 0.6335770487785339, "rewards/total_composite/std": 0.19988654553890228, "reward": 0.6335770487785339, "reward_std": 0.19988654553890228, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21457158029079437, "sampling/sampling_logp_difference/max": 2.245924949645996, "sampling/importance_sampling_ratio/min": 0.10582961142063141, "sampling/importance_sampling_ratio/mean": 1.0353453159332275, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.466738671064377, "clip_ratio/low_mean": 0.09431399963796139, "clip_ratio/low_min": 0.09431399963796139, "clip_ratio/high_mean": 0.10586324706673622, "clip_ratio/high_max": 0.10586324706673622, "clip_ratio/region_mean": 0.2001772467046976, "reward_total_mean": 0.6335770487785339, "reward_meter_mean": 0.5344254970550537, "reward_meter_std": 0.3820936977863312, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.99835604429245, "reward_repeat_soft_std": 0.0023890030570328236, "reward_judge_quality_mean": 0.47749996185302734, "reward_judge_quality_std": 0.16263456642627716, "reward_total_composite_mean": 0.6335770487785339, "reward_total_composite_std": 0.19988654553890228} {"timestamp_utc": "2026-04-12T22:30:39Z", "mode": "train", "global_step": 116, "epoch": 0.011652435961828227, "loss": 0.0721, "grad_norm": 14.912810325622559, "learning_rate": 9.651515151515153e-06, "num_tokens": 220027.0, "completions/mean_length": 47.875, "completions/min_length": 41.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.875, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.5695972442626953, "rewards/meter/std": 0.41784724593162537, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9951172471046448, "rewards/repeat_soft/std": 0.007368812803179026, "rewards/judge_quality/mean": 0.6987500190734863, "rewards/judge_quality/std": 0.24474114179611206, "rewards/total_composite/mean": 0.7154554128646851, "rewards/total_composite/std": 0.19057126343250275, "reward": 0.7154554128646851, "reward_std": 0.19057126343250275, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17878463864326477, "sampling/sampling_logp_difference/max": 1.5432236194610596, "sampling/importance_sampling_ratio/min": 0.2136911302804947, "sampling/importance_sampling_ratio/mean": 1.0183104276657104, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.463243305683136, "clip_ratio/low_mean": 0.08487668633460999, "clip_ratio/low_min": 0.08487668633460999, "clip_ratio/high_mean": 0.1055582882836461, "clip_ratio/high_max": 0.1055582882836461, "clip_ratio/region_mean": 0.1904349746182561, "reward_total_mean": 0.7154554128646851, "reward_meter_mean": 0.5695972442626953, "reward_meter_std": 0.41784724593162537, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9951172471046448, "reward_repeat_soft_std": 0.007368812803179026, "reward_judge_quality_mean": 0.6987500190734863, "reward_judge_quality_std": 0.24474114179611206, "reward_total_composite_mean": 0.7154554128646851, "reward_total_composite_std": 0.19057126343250275} {"timestamp_utc": "2026-04-12T22:30:45Z", "mode": "train", "global_step": 117, "epoch": 0.011752887995981919, "loss": 0.013, "grad_norm": 19.91445541381836, "learning_rate": 9.648484848484849e-06, "num_tokens": 221624.0, "completions/mean_length": 48.625, "completions/min_length": 36.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.625, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9055629372596741, "rewards/meter/std": 0.1779366135597229, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9891988039016724, "rewards/repeat_soft/std": 0.002124833408743143, "rewards/judge_quality/mean": 0.8650000095367432, "rewards/judge_quality/std": 0.09055384993553162, "rewards/total_composite/mean": 0.9159232378005981, "rewards/total_composite/std": 0.1001143530011177, "reward": 0.9159232378005981, "reward_std": 0.10011434555053711, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08398713171482086, "sampling/sampling_logp_difference/max": 1.432332992553711, "sampling/importance_sampling_ratio/min": 0.23875127732753754, "sampling/importance_sampling_ratio/mean": 0.9926836490631104, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.28090511448681355, "clip_ratio/low_mean": 0.022110641933977604, "clip_ratio/low_min": 0.022110641933977604, "clip_ratio/high_mean": 0.039251322858035564, "clip_ratio/high_max": 0.039251322858035564, "clip_ratio/region_mean": 0.06136196479201317, "reward_total_mean": 0.9159232378005981, "reward_meter_mean": 0.9055629372596741, "reward_meter_std": 0.1779366135597229, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9891988039016724, "reward_repeat_soft_std": 0.002124833408743143, "reward_judge_quality_mean": 0.8650000095367432, "reward_judge_quality_std": 0.09055384993553162, "reward_total_composite_mean": 0.9159232378005981, "reward_total_composite_std": 0.1001143530011177} {"timestamp_utc": "2026-04-12T22:30:53Z", "mode": "train", "global_step": 118, "epoch": 0.01185334003013561, "loss": -0.0515, "grad_norm": 7.082086086273193, "learning_rate": 9.645454545454548e-06, "num_tokens": 224495.0, "completions/mean_length": 158.875, "completions/min_length": 130.0, "completions/max_length": 180.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 158.875, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 180.0, "rewards/meter/mean": 0.7066109776496887, "rewards/meter/std": 0.28142380714416504, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9959703087806702, "rewards/repeat_soft/std": 0.0038161210250109434, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.631685733795166, "rewards/total_composite/std": 0.27332279086112976, "reward": 0.631685733795166, "reward_std": 0.27332279086112976, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20089176297187805, "sampling/sampling_logp_difference/max": 1.853123664855957, "sampling/importance_sampling_ratio/min": 0.1567467898130417, "sampling/importance_sampling_ratio/mean": 1.0425838232040405, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4911645650863647, "clip_ratio/low_mean": 0.07768536917865276, "clip_ratio/low_min": 0.07768536917865276, "clip_ratio/high_mean": 0.12430565804243088, "clip_ratio/high_max": 0.12430565804243088, "clip_ratio/region_mean": 0.20199102722108364, "reward_total_mean": 0.631685733795166, "reward_meter_mean": 0.7066109776496887, "reward_meter_std": 0.28142380714416504, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9959703087806702, "reward_repeat_soft_std": 0.0038161210250109434, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.631685733795166, "reward_total_composite_std": 0.27332279086112976} {"timestamp_utc": "2026-04-12T22:31:00Z", "mode": "train", "global_step": 119, "epoch": 0.011953792064289303, "loss": -0.0428, "grad_norm": 12.962621688842773, "learning_rate": 9.642424242424243e-06, "num_tokens": 226327.0, "completions/mean_length": 63.0, "completions/min_length": 44.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.7072560787200928, "rewards/meter/std": 0.3508759140968323, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.996657133102417, "rewards/repeat_soft/std": 0.0072163003496825695, "rewards/judge_quality/mean": 0.45625001192092896, "rewards/judge_quality/std": 0.21185828745365143, "rewards/total_composite/mean": 0.6302478313446045, "rewards/total_composite/std": 0.31918975710868835, "reward": 0.6302478313446045, "reward_std": 0.31918975710868835, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1847275048494339, "sampling/sampling_logp_difference/max": 1.2999019622802734, "sampling/importance_sampling_ratio/min": 0.2725585103034973, "sampling/importance_sampling_ratio/mean": 1.0211437940597534, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0329722613096237, "clip_ratio/low_mean": 0.05912642180919647, "clip_ratio/low_min": 0.05912642180919647, "clip_ratio/high_mean": 0.12136753089725971, "clip_ratio/high_max": 0.12136753089725971, "clip_ratio/region_mean": 0.18049395270645618, "reward_total_mean": 0.6302478313446045, "reward_meter_mean": 0.7072560787200928, "reward_meter_std": 0.3508759140968323, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.996657133102417, "reward_repeat_soft_std": 0.0072163003496825695, "reward_judge_quality_mean": 0.45625001192092896, "reward_judge_quality_std": 0.21185828745365143, "reward_total_composite_mean": 0.6302478313446045, "reward_total_composite_std": 0.31918975710868835} {"timestamp_utc": "2026-04-12T22:31:11Z", "mode": "train", "global_step": 120, "epoch": 0.012054244098442994, "loss": -0.1991, "grad_norm": 3.504687547683716, "learning_rate": 9.63939393939394e-06, "num_tokens": 228691.0, "completions/mean_length": 165.5, "completions/min_length": 94.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 116.00000762939453, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.2696044445037842, "rewards/meter/std": 0.2580581307411194, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9958187341690063, "rewards/repeat_soft/std": 0.003901544027030468, "rewards/judge_quality/mean": 0.398749977350235, "rewards/judge_quality/std": 0.17125065624713898, "rewards/total_composite/mean": 0.4501357078552246, "rewards/total_composite/std": 0.20900095999240875, "reward": 0.4501357078552246, "reward_std": 0.20900094509124756, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1954592615365982, "sampling/sampling_logp_difference/max": 1.9527530670166016, "sampling/importance_sampling_ratio/min": 0.14188291132450104, "sampling/importance_sampling_ratio/mean": 1.022305965423584, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7254022061824799, "clip_ratio/low_mean": 0.04175109788775444, "clip_ratio/low_min": 0.04175109788775444, "clip_ratio/high_mean": 0.13687382265925407, "clip_ratio/high_max": 0.13687382265925407, "clip_ratio/region_mean": 0.17862492054700851, "reward_total_mean": 0.4501357078552246, "reward_meter_mean": 0.2696044445037842, "reward_meter_std": 0.2580581307411194, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9958187341690063, "reward_repeat_soft_std": 0.003901544027030468, "reward_judge_quality_mean": 0.398749977350235, "reward_judge_quality_std": 0.17125065624713898, "reward_total_composite_mean": 0.4501357078552246, "reward_total_composite_std": 0.20900095999240875} {"timestamp_utc": "2026-04-12T22:31:20Z", "mode": "train", "global_step": 121, "epoch": 0.012154696132596685, "loss": 0.0322, "grad_norm": 8.148179054260254, "learning_rate": 9.636363636363638e-06, "num_tokens": 231502.0, "completions/mean_length": 151.375, "completions/min_length": 122.0, "completions/max_length": 171.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 151.375, "completions/min_terminated_length": 122.0, "completions/max_terminated_length": 171.0, "rewards/meter/mean": 0.46986302733421326, "rewards/meter/std": 0.25895586609840393, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9900440573692322, "rewards/repeat_soft/std": 0.007098556496202946, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.5838177800178528, "rewards/total_composite/std": 0.12632137537002563, "reward": 0.5838177800178528, "reward_std": 0.12632139027118683, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20250819623470306, "sampling/sampling_logp_difference/max": 1.9475808143615723, "sampling/importance_sampling_ratio/min": 0.14261867105960846, "sampling/importance_sampling_ratio/mean": 1.0346516370773315, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.129692703485489, "clip_ratio/low_mean": 0.10343983769416809, "clip_ratio/low_min": 0.10343983769416809, "clip_ratio/high_mean": 0.10095386765897274, "clip_ratio/high_max": 0.10095386765897274, "clip_ratio/region_mean": 0.20439370535314083, "reward_total_mean": 0.5838177800178528, "reward_meter_mean": 0.46986302733421326, "reward_meter_std": 0.25895586609840393, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9900440573692322, "reward_repeat_soft_std": 0.007098556496202946, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.5838177800178528, "reward_total_composite_std": 0.12632137537002563} {"timestamp_utc": "2026-04-12T22:31:26Z", "mode": "train", "global_step": 122, "epoch": 0.012255148166750376, "loss": -0.0128, "grad_norm": 10.772955894470215, "learning_rate": 9.633333333333335e-06, "num_tokens": 233265.0, "completions/mean_length": 62.375, "completions/min_length": 49.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.375, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.840729832649231, "rewards/meter/std": 0.21353866159915924, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9988186359405518, "rewards/repeat_soft/std": 0.0025750070344656706, "rewards/judge_quality/mean": 0.5175000429153442, "rewards/judge_quality/std": 0.12498573213815689, "rewards/total_composite/mean": 0.5670230984687805, "rewards/total_composite/std": 0.36396607756614685, "reward": 0.5670230984687805, "reward_std": 0.36396604776382446, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21837672591209412, "sampling/sampling_logp_difference/max": 1.421654224395752, "sampling/importance_sampling_ratio/min": 0.24131451547145844, "sampling/importance_sampling_ratio/mean": 1.0179781913757324, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.149739980697632, "clip_ratio/low_mean": 0.046270232647657394, "clip_ratio/low_min": 0.046270232647657394, "clip_ratio/high_mean": 0.15077135432511568, "clip_ratio/high_max": 0.15077135432511568, "clip_ratio/region_mean": 0.19704158697277308, "reward_total_mean": 0.5670230984687805, "reward_meter_mean": 0.840729832649231, "reward_meter_std": 0.21353866159915924, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9988186359405518, "reward_repeat_soft_std": 0.0025750070344656706, "reward_judge_quality_mean": 0.5175000429153442, "reward_judge_quality_std": 0.12498573213815689, "reward_total_composite_mean": 0.5670230984687805, "reward_total_composite_std": 0.36396607756614685} {"timestamp_utc": "2026-04-12T22:31:32Z", "mode": "train", "global_step": 123, "epoch": 0.012355600200904069, "loss": -0.0751, "grad_norm": 19.844261169433594, "learning_rate": 9.63030303030303e-06, "num_tokens": 234744.0, "completions/mean_length": 36.875, "completions/min_length": 26.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.875, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.154674232006073, "rewards/meter/std": 0.19714823365211487, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9931535720825195, "rewards/repeat_soft/std": 0.00823448970913887, "rewards/judge_quality/mean": 0.6525000333786011, "rewards/judge_quality/std": 0.2921227812767029, "rewards/total_composite/mean": 0.47386685013771057, "rewards/total_composite/std": 0.22027365863323212, "reward": 0.47386685013771057, "reward_std": 0.22027365863323212, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19463254511356354, "sampling/sampling_logp_difference/max": 1.9416074752807617, "sampling/importance_sampling_ratio/min": 0.14347313344478607, "sampling/importance_sampling_ratio/mean": 1.0010665655136108, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0737264752388, "clip_ratio/low_mean": 0.0697358213365078, "clip_ratio/low_min": 0.0697358213365078, "clip_ratio/high_mean": 0.08342572208493948, "clip_ratio/high_max": 0.08342572208493948, "clip_ratio/region_mean": 0.15316154342144728, "reward_total_mean": 0.47386685013771057, "reward_meter_mean": 0.154674232006073, "reward_meter_std": 0.19714823365211487, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9931535720825195, "reward_repeat_soft_std": 0.00823448970913887, "reward_judge_quality_mean": 0.6525000333786011, "reward_judge_quality_std": 0.2921227812767029, "reward_total_composite_mean": 0.47386685013771057, "reward_total_composite_std": 0.22027365863323212} {"timestamp_utc": "2026-04-12T22:31:39Z", "mode": "train", "global_step": 124, "epoch": 0.01245605223505776, "loss": 0.0995, "grad_norm": 22.603771209716797, "learning_rate": 9.627272727272728e-06, "num_tokens": 236144.0, "completions/mean_length": 26.0, "completions/min_length": 18.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.0, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.7815031409263611, "rewards/meter/std": 0.372760534286499, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.959294855594635, "rewards/repeat_soft/std": 0.009065471589565277, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.7446058988571167, "rewards/total_composite/std": 0.18518446385860443, "reward": 0.7446058988571167, "reward_std": 0.18518446385860443, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18050359189510345, "sampling/sampling_logp_difference/max": 1.3536615371704102, "sampling/importance_sampling_ratio/min": 0.2747851610183716, "sampling/importance_sampling_ratio/mean": 1.0089120864868164, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.512794516980648, "clip_ratio/low_mean": 0.03327047452330589, "clip_ratio/low_min": 0.03327047452330589, "clip_ratio/high_mean": 0.14881853945553303, "clip_ratio/high_max": 0.14881853945553303, "clip_ratio/region_mean": 0.18208901397883892, "reward_total_mean": 0.7446058988571167, "reward_meter_mean": 0.7815031409263611, "reward_meter_std": 0.372760534286499, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.959294855594635, "reward_repeat_soft_std": 0.009065471589565277, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.7446058988571167, "reward_total_composite_std": 0.18518446385860443} {"timestamp_utc": "2026-04-12T22:31:46Z", "mode": "train", "global_step": 125, "epoch": 0.012556504269211451, "loss": 0.0231, "grad_norm": 18.275697708129883, "learning_rate": 9.624242424242425e-06, "num_tokens": 238133.0, "completions/mean_length": 79.625, "completions/min_length": 64.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.625, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.4361572563648224, "rewards/meter/std": 0.2553798258304596, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9925416707992554, "rewards/repeat_soft/std": 0.006211057770997286, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.5389169454574585, "rewards/total_composite/std": 0.24676160514354706, "reward": 0.5389169454574585, "reward_std": 0.24676157534122467, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22753405570983887, "sampling/sampling_logp_difference/max": 2.1534345149993896, "sampling/importance_sampling_ratio/min": 0.11608477681875229, "sampling/importance_sampling_ratio/mean": 1.0170170068740845, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.442027322947979, "clip_ratio/low_mean": 0.05522531643509865, "clip_ratio/low_min": 0.05522531643509865, "clip_ratio/high_mean": 0.15537106804549694, "clip_ratio/high_max": 0.15537106804549694, "clip_ratio/region_mean": 0.2105963844805956, "reward_total_mean": 0.5389169454574585, "reward_meter_mean": 0.4361572563648224, "reward_meter_std": 0.2553798258304596, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9925416707992554, "reward_repeat_soft_std": 0.006211057770997286, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.5389169454574585, "reward_total_composite_std": 0.24676160514354706} {"timestamp_utc": "2026-04-12T22:31:51Z", "mode": "train", "global_step": 126, "epoch": 0.012656956303365142, "loss": 0.0736, "grad_norm": 14.959473609924316, "learning_rate": 9.621212121212122e-06, "num_tokens": 239531.0, "completions/mean_length": 30.75, "completions/min_length": 26.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.75, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9794123768806458, "rewards/meter/std": 0.026296310126781464, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9514520168304443, "rewards/repeat_soft/std": 0.01740238256752491, "rewards/judge_quality/mean": 0.5400000214576721, "rewards/judge_quality/std": 0.33393755555152893, "rewards/total_composite/mean": 0.8478807210922241, "rewards/total_composite/std": 0.10403254628181458, "reward": 0.8478807210922241, "reward_std": 0.10403254628181458, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1731196641921997, "sampling/sampling_logp_difference/max": 1.1950368881225586, "sampling/importance_sampling_ratio/min": 0.3026927709579468, "sampling/importance_sampling_ratio/mean": 1.0100114345550537, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6873226910829544, "clip_ratio/low_mean": 0.1017175568267703, "clip_ratio/low_min": 0.1017175568267703, "clip_ratio/high_mean": 0.05982143059372902, "clip_ratio/high_max": 0.05982143059372902, "clip_ratio/region_mean": 0.16153898742049932, "reward_total_mean": 0.8478807210922241, "reward_meter_mean": 0.9794123768806458, "reward_meter_std": 0.026296310126781464, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9514520168304443, "reward_repeat_soft_std": 0.01740238256752491, "reward_judge_quality_mean": 0.5400000214576721, "reward_judge_quality_std": 0.33393755555152893, "reward_total_composite_mean": 0.8478807210922241, "reward_total_composite_std": 0.10403254628181458} {"timestamp_utc": "2026-04-12T22:31:58Z", "mode": "train", "global_step": 127, "epoch": 0.012757408337518835, "loss": 0.0008, "grad_norm": 13.659857749938965, "learning_rate": 9.61818181818182e-06, "num_tokens": 241182.0, "completions/mean_length": 60.375, "completions/min_length": 53.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.375, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.6842314004898071, "rewards/meter/std": 0.4053438603878021, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 1.0, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.6797791123390198, "rewards/total_composite/std": 0.17921513319015503, "reward": 0.6797791123390198, "reward_std": 0.17921513319015503, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20752720534801483, "sampling/sampling_logp_difference/max": 1.3170757293701172, "sampling/importance_sampling_ratio/min": 0.2679176330566406, "sampling/importance_sampling_ratio/mean": 1.0146310329437256, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.952474370598793, "clip_ratio/low_mean": 0.052476415410637856, "clip_ratio/low_min": 0.052476415410637856, "clip_ratio/high_mean": 0.17280091531574726, "clip_ratio/high_max": 0.17280091531574726, "clip_ratio/region_mean": 0.22527733072638512, "reward_total_mean": 0.6797791123390198, "reward_meter_mean": 0.6842314004898071, "reward_meter_std": 0.4053438603878021, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 1.0, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.6797791123390198, "reward_total_composite_std": 0.17921513319015503} {"timestamp_utc": "2026-04-12T22:32:05Z", "mode": "train", "global_step": 128, "epoch": 0.012857860371672527, "loss": 0.1126, "grad_norm": 9.270278930664062, "learning_rate": 9.615151515151517e-06, "num_tokens": 243477.0, "completions/mean_length": 113.875, "completions/min_length": 93.0, "completions/max_length": 149.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.875, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 149.0, "rewards/meter/mean": 0.36527761816978455, "rewards/meter/std": 0.41165366768836975, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9943387508392334, "rewards/repeat_soft/std": 0.004277929663658142, "rewards/judge_quality/mean": 0.5099999904632568, "rewards/judge_quality/std": 0.2343989461660385, "rewards/total_composite/mean": 0.5621213316917419, "rewards/total_composite/std": 0.23888105154037476, "reward": 0.5621213316917419, "reward_std": 0.23888105154037476, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2154771238565445, "sampling/sampling_logp_difference/max": 2.0755271911621094, "sampling/importance_sampling_ratio/min": 0.12549026310443878, "sampling/importance_sampling_ratio/mean": 1.047012448310852, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.288969650864601, "clip_ratio/low_mean": 0.10721415467560291, "clip_ratio/low_min": 0.10721415467560291, "clip_ratio/high_mean": 0.08637929800897837, "clip_ratio/high_max": 0.08637929800897837, "clip_ratio/region_mean": 0.19359345268458128, "reward_total_mean": 0.5621213316917419, "reward_meter_mean": 0.36527761816978455, "reward_meter_std": 0.41165366768836975, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9943387508392334, "reward_repeat_soft_std": 0.004277929663658142, "reward_judge_quality_mean": 0.5099999904632568, "reward_judge_quality_std": 0.2343989461660385, "reward_total_composite_mean": 0.5621213316917419, "reward_total_composite_std": 0.23888105154037476} {"timestamp_utc": "2026-04-12T22:32:13Z", "mode": "train", "global_step": 129, "epoch": 0.012958312405826218, "loss": 0.0925, "grad_norm": 10.018613815307617, "learning_rate": 9.612121212121212e-06, "num_tokens": 245700.0, "completions/mean_length": 124.875, "completions/min_length": 76.0, "completions/max_length": 144.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 124.875, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 144.0, "rewards/meter/mean": 0.8152534365653992, "rewards/meter/std": 0.2667849361896515, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.995792806148529, "rewards/repeat_soft/std": 0.003435641061514616, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7250058650970459, "rewards/total_composite/std": 0.13674314320087433, "reward": 0.7250058650970459, "reward_std": 0.13674314320087433, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19991059601306915, "sampling/sampling_logp_difference/max": 1.490544319152832, "sampling/importance_sampling_ratio/min": 0.2252500206232071, "sampling/importance_sampling_ratio/mean": 1.0232343673706055, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2211784571409225, "clip_ratio/low_mean": 0.049293627962470055, "clip_ratio/low_min": 0.049293627962470055, "clip_ratio/high_mean": 0.16842706594616175, "clip_ratio/high_max": 0.16842706594616175, "clip_ratio/region_mean": 0.2177206939086318, "reward_total_mean": 0.7250058650970459, "reward_meter_mean": 0.8152534365653992, "reward_meter_std": 0.2667849361896515, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.995792806148529, "reward_repeat_soft_std": 0.003435641061514616, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7250058650970459, "reward_total_composite_std": 0.13674314320087433} {"timestamp_utc": "2026-04-12T22:32:20Z", "mode": "train", "global_step": 130, "epoch": 0.013058764439979909, "loss": 0.0572, "grad_norm": 8.2088041305542, "learning_rate": 9.60909090909091e-06, "num_tokens": 248142.0, "completions/mean_length": 128.25, "completions/min_length": 89.0, "completions/max_length": 164.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.25, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 164.0, "rewards/meter/mean": 0.7240664958953857, "rewards/meter/std": 0.30356472730636597, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9945056438446045, "rewards/repeat_soft/std": 0.0045688156969845295, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.6774680018424988, "rewards/total_composite/std": 0.1481560319662094, "reward": 0.6774680018424988, "reward_std": 0.1481560319662094, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20742681622505188, "sampling/sampling_logp_difference/max": 1.6332435607910156, "sampling/importance_sampling_ratio/min": 0.1952950805425644, "sampling/importance_sampling_ratio/mean": 1.0403975248336792, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.306053400039673, "clip_ratio/low_mean": 0.059869431890547276, "clip_ratio/low_min": 0.059869431890547276, "clip_ratio/high_mean": 0.12077674269676208, "clip_ratio/high_max": 0.12077674269676208, "clip_ratio/region_mean": 0.18064617458730936, "reward_total_mean": 0.6774680018424988, "reward_meter_mean": 0.7240664958953857, "reward_meter_std": 0.30356472730636597, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9945056438446045, "reward_repeat_soft_std": 0.0045688156969845295, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.6774680018424988, "reward_total_composite_std": 0.1481560319662094} {"timestamp_utc": "2026-04-12T22:32:26Z", "mode": "train", "global_step": 131, "epoch": 0.013159216474133602, "loss": -0.0276, "grad_norm": 16.39191246032715, "learning_rate": 9.606060606060607e-06, "num_tokens": 249709.0, "completions/mean_length": 37.875, "completions/min_length": 28.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.875, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.6100648641586304, "rewards/meter/std": 0.42265385389328003, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9827827215194702, "rewards/repeat_soft/std": 0.029818646609783173, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.2499571591615677, "rewards/total_composite/mean": 0.6810574531555176, "rewards/total_composite/std": 0.15360815823078156, "reward": 0.6810574531555176, "reward_std": 0.15360815823078156, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2118445187807083, "sampling/sampling_logp_difference/max": 2.0108089447021484, "sampling/importance_sampling_ratio/min": 0.13388033211231232, "sampling/importance_sampling_ratio/mean": 1.0341925621032715, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3326124921441078, "clip_ratio/low_mean": 0.11372986994683743, "clip_ratio/low_min": 0.11372986994683743, "clip_ratio/high_mean": 0.14517483115196228, "clip_ratio/high_max": 0.14517483115196228, "clip_ratio/region_mean": 0.2589047010987997, "reward_total_mean": 0.6810574531555176, "reward_meter_mean": 0.6100648641586304, "reward_meter_std": 0.42265385389328003, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9827827215194702, "reward_repeat_soft_std": 0.029818646609783173, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.2499571591615677, "reward_total_composite_mean": 0.6810574531555176, "reward_total_composite_std": 0.15360815823078156} {"timestamp_utc": "2026-04-12T22:32:33Z", "mode": "train", "global_step": 132, "epoch": 0.013259668508287293, "loss": 0.0561, "grad_norm": 8.524138450622559, "learning_rate": 9.603030303030304e-06, "num_tokens": 252444.0, "completions/mean_length": 138.875, "completions/min_length": 127.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 138.875, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.3716757893562317, "rewards/meter/std": 0.3261376619338989, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9943193793296814, "rewards/repeat_soft/std": 0.0035428176634013653, "rewards/judge_quality/mean": 0.39375001192092896, "rewards/judge_quality/std": 0.15638209879398346, "rewards/total_composite/mean": 0.5348110198974609, "rewards/total_composite/std": 0.18445618450641632, "reward": 0.5348110198974609, "reward_std": 0.18445616960525513, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18236319720745087, "sampling/sampling_logp_difference/max": 1.5875434875488281, "sampling/importance_sampling_ratio/min": 0.20442716777324677, "sampling/importance_sampling_ratio/mean": 1.0325572490692139, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9220962226390839, "clip_ratio/low_mean": 0.09894000738859177, "clip_ratio/low_min": 0.09894000738859177, "clip_ratio/high_mean": 0.07793621346354485, "clip_ratio/high_max": 0.07793621346354485, "clip_ratio/region_mean": 0.1768762208521366, "reward_total_mean": 0.5348110198974609, "reward_meter_mean": 0.3716757893562317, "reward_meter_std": 0.3261376619338989, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9943193793296814, "reward_repeat_soft_std": 0.0035428176634013653, "reward_judge_quality_mean": 0.39375001192092896, "reward_judge_quality_std": 0.15638209879398346, "reward_total_composite_mean": 0.5348110198974609, "reward_total_composite_std": 0.18445618450641632} {"timestamp_utc": "2026-04-12T22:32:41Z", "mode": "train", "global_step": 133, "epoch": 0.013360120542440984, "loss": 0.1566, "grad_norm": 18.800283432006836, "learning_rate": 9.600000000000001e-06, "num_tokens": 254891.0, "completions/mean_length": 106.875, "completions/min_length": 84.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.875, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.9262809753417969, "rewards/meter/std": 0.07574860751628876, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.24800792336463928, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9938892126083374, "rewards/repeat_soft/std": 0.008599473163485527, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7745903730392456, "rewards/total_composite/std": 0.05820745229721069, "reward": 0.7745903730392456, "reward_std": 0.05820745974779129, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18495529890060425, "sampling/sampling_logp_difference/max": 1.525949478149414, "sampling/importance_sampling_ratio/min": 0.21741452813148499, "sampling/importance_sampling_ratio/mean": 1.0175228118896484, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.949052482843399, "clip_ratio/low_mean": 0.0336101409047842, "clip_ratio/low_min": 0.0336101409047842, "clip_ratio/high_mean": 0.145680645480752, "clip_ratio/high_max": 0.145680645480752, "clip_ratio/region_mean": 0.1792907863855362, "reward_total_mean": 0.7745903730392456, "reward_meter_mean": 0.9262809753417969, "reward_meter_std": 0.07574860751628876, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.24800792336463928, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9938892126083374, "reward_repeat_soft_std": 0.008599473163485527, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7745903730392456, "reward_total_composite_std": 0.05820745229721069} {"timestamp_utc": "2026-04-12T22:32:49Z", "mode": "train", "global_step": 134, "epoch": 0.013460572576594675, "loss": 0.0607, "grad_norm": 7.037054538726807, "learning_rate": 9.596969696969699e-06, "num_tokens": 257734.0, "completions/mean_length": 162.375, "completions/min_length": 123.0, "completions/max_length": 186.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 162.375, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 186.0, "rewards/meter/mean": 0.9117918014526367, "rewards/meter/std": 0.22455473244190216, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9934555292129517, "rewards/repeat_soft/std": 0.005569420754909515, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.14520922303199768, "rewards/total_composite/mean": 0.587673008441925, "rewards/total_composite/std": 0.3813877999782562, "reward": 0.587673008441925, "reward_std": 0.38138777017593384, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19455531239509583, "sampling/sampling_logp_difference/max": 1.505533218383789, "sampling/importance_sampling_ratio/min": 0.22189894318580627, "sampling/importance_sampling_ratio/mean": 1.0390830039978027, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4518264532089233, "clip_ratio/low_mean": 0.0657502580434084, "clip_ratio/low_min": 0.0657502580434084, "clip_ratio/high_mean": 0.12715561501681805, "clip_ratio/high_max": 0.12715561501681805, "clip_ratio/region_mean": 0.19290587306022644, "reward_total_mean": 0.587673008441925, "reward_meter_mean": 0.9117918014526367, "reward_meter_std": 0.22455473244190216, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9934555292129517, "reward_repeat_soft_std": 0.005569420754909515, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.14520922303199768, "reward_total_composite_mean": 0.587673008441925, "reward_total_composite_std": 0.3813877999782562} {"timestamp_utc": "2026-04-12T22:32:56Z", "mode": "train", "global_step": 135, "epoch": 0.013561024610748368, "loss": 0.0783, "grad_norm": 7.698927879333496, "learning_rate": 9.593939393939394e-06, "num_tokens": 260125.0, "completions/mean_length": 132.875, "completions/min_length": 86.0, "completions/max_length": 170.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.875, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 170.0, "rewards/meter/mean": 0.512225329875946, "rewards/meter/std": 0.2971871495246887, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9965784549713135, "rewards/repeat_soft/std": 0.003520975122228265, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.22403764724731445, "rewards/total_composite/mean": 0.5387158393859863, "rewards/total_composite/std": 0.27052682638168335, "reward": 0.5387158393859863, "reward_std": 0.27052679657936096, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20126554369926453, "sampling/sampling_logp_difference/max": 1.7666950225830078, "sampling/importance_sampling_ratio/min": 0.17089687287807465, "sampling/importance_sampling_ratio/mean": 1.0329970121383667, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3159025609493256, "clip_ratio/low_mean": 0.0917290709912777, "clip_ratio/low_min": 0.0917290709912777, "clip_ratio/high_mean": 0.11301391199231148, "clip_ratio/high_max": 0.11301391199231148, "clip_ratio/region_mean": 0.20474298298358917, "reward_total_mean": 0.5387158393859863, "reward_meter_mean": 0.512225329875946, "reward_meter_std": 0.2971871495246887, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9965784549713135, "reward_repeat_soft_std": 0.003520975122228265, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.22403764724731445, "reward_total_composite_mean": 0.5387158393859863, "reward_total_composite_std": 0.27052682638168335} {"timestamp_utc": "2026-04-12T22:33:03Z", "mode": "train", "global_step": 136, "epoch": 0.01366147664490206, "loss": 0.0376, "grad_norm": 9.772281646728516, "learning_rate": 9.590909090909091e-06, "num_tokens": 262290.0, "completions/mean_length": 90.625, "completions/min_length": 76.0, "completions/max_length": 119.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.625, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.5402966737747192, "rewards/meter/std": 0.4211829900741577, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.984871506690979, "rewards/repeat_soft/std": 0.013639903627336025, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.6112963557243347, "rewards/total_composite/std": 0.3185757100582123, "reward": 0.6112963557243347, "reward_std": 0.3185757100582123, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20179298520088196, "sampling/sampling_logp_difference/max": 1.3335742950439453, "sampling/importance_sampling_ratio/min": 0.26353365182876587, "sampling/importance_sampling_ratio/mean": 0.9979775547981262, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.586782455444336, "clip_ratio/low_mean": 0.05890737473964691, "clip_ratio/low_min": 0.05890737473964691, "clip_ratio/high_mean": 0.1278254184871912, "clip_ratio/high_max": 0.1278254184871912, "clip_ratio/region_mean": 0.1867327932268381, "reward_total_mean": 0.6112963557243347, "reward_meter_mean": 0.5402966737747192, "reward_meter_std": 0.4211829900741577, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.984871506690979, "reward_repeat_soft_std": 0.013639903627336025, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.6112963557243347, "reward_total_composite_std": 0.3185757100582123} {"timestamp_utc": "2026-04-12T22:33:09Z", "mode": "train", "global_step": 137, "epoch": 0.01376192867905575, "loss": -0.0179, "grad_norm": 15.270221710205078, "learning_rate": 9.587878787878789e-06, "num_tokens": 263927.0, "completions/mean_length": 50.625, "completions/min_length": 38.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.625, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.12983430922031403, "rewards/meter/std": 0.1328689157962799, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9958950877189636, "rewards/repeat_soft/std": 0.005419950000941753, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.2499571591615677, "rewards/total_composite/mean": 0.37045469880104065, "rewards/total_composite/std": 0.23874065279960632, "reward": 0.37045469880104065, "reward_std": 0.23874063789844513, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20379044115543365, "sampling/sampling_logp_difference/max": 1.362325668334961, "sampling/importance_sampling_ratio/min": 0.2560645639896393, "sampling/importance_sampling_ratio/mean": 1.0346601009368896, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0476372092962265, "clip_ratio/low_mean": 0.05046296305954456, "clip_ratio/low_min": 0.05046296305954456, "clip_ratio/high_mean": 0.1526920609176159, "clip_ratio/high_max": 0.1526920609176159, "clip_ratio/region_mean": 0.20315502397716045, "reward_total_mean": 0.37045469880104065, "reward_meter_mean": 0.12983430922031403, "reward_meter_std": 0.1328689157962799, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9958950877189636, "reward_repeat_soft_std": 0.005419950000941753, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.2499571591615677, "reward_total_composite_mean": 0.37045469880104065, "reward_total_composite_std": 0.23874065279960632} {"timestamp_utc": "2026-04-12T22:33:15Z", "mode": "train", "global_step": 138, "epoch": 0.013862380713209442, "loss": -0.039, "grad_norm": 17.487319946289062, "learning_rate": 9.584848484848486e-06, "num_tokens": 265676.0, "completions/mean_length": 49.625, "completions/min_length": 43.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.625, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.378584086894989, "rewards/meter/std": 0.35492274165153503, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9902465343475342, "rewards/repeat_soft/std": 0.011248313821852207, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.5539078712463379, "rewards/total_composite/std": 0.26821252703666687, "reward": 0.5539078712463379, "reward_std": 0.26821252703666687, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2223401814699173, "sampling/sampling_logp_difference/max": 2.262995719909668, "sampling/importance_sampling_ratio/min": 0.10403835028409958, "sampling/importance_sampling_ratio/mean": 1.0410206317901611, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.078471839427948, "clip_ratio/low_mean": 0.12089350260794163, "clip_ratio/low_min": 0.12089350260794163, "clip_ratio/high_mean": 0.1181613514199853, "clip_ratio/high_max": 0.1181613514199853, "clip_ratio/region_mean": 0.23905485402792692, "reward_total_mean": 0.5539078712463379, "reward_meter_mean": 0.378584086894989, "reward_meter_std": 0.35492274165153503, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9902465343475342, "reward_repeat_soft_std": 0.011248313821852207, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.5539078712463379, "reward_total_composite_std": 0.26821252703666687} {"timestamp_utc": "2026-04-12T22:33:22Z", "mode": "train", "global_step": 139, "epoch": 0.013962832747363135, "loss": 0.0251, "grad_norm": 10.91278076171875, "learning_rate": 9.581818181818181e-06, "num_tokens": 267487.0, "completions/mean_length": 75.375, "completions/min_length": 63.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.375, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.7106475830078125, "rewards/meter/std": 0.3840155303478241, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9995355010032654, "rewards/repeat_soft/std": 0.000715656962711364, "rewards/judge_quality/mean": 0.4674999713897705, "rewards/judge_quality/std": 0.179344043135643, "rewards/total_composite/mean": 0.7006199955940247, "rewards/total_composite/std": 0.19790136814117432, "reward": 0.7006199955940247, "reward_std": 0.1979013830423355, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1969791203737259, "sampling/sampling_logp_difference/max": 1.579707145690918, "sampling/importance_sampling_ratio/min": 0.20603543519973755, "sampling/importance_sampling_ratio/mean": 1.0220390558242798, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9457141309976578, "clip_ratio/low_mean": 0.0600130520761013, "clip_ratio/low_min": 0.0600130520761013, "clip_ratio/high_mean": 0.1256153592839837, "clip_ratio/high_max": 0.1256153592839837, "clip_ratio/region_mean": 0.185628411360085, "reward_total_mean": 0.7006199955940247, "reward_meter_mean": 0.7106475830078125, "reward_meter_std": 0.3840155303478241, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9995355010032654, "reward_repeat_soft_std": 0.000715656962711364, "reward_judge_quality_mean": 0.4674999713897705, "reward_judge_quality_std": 0.179344043135643, "reward_total_composite_mean": 0.7006199955940247, "reward_total_composite_std": 0.19790136814117432} {"timestamp_utc": "2026-04-12T22:33:28Z", "mode": "train", "global_step": 140, "epoch": 0.014063284781516826, "loss": -0.0163, "grad_norm": 24.94887924194336, "learning_rate": 9.57878787878788e-06, "num_tokens": 269104.0, "completions/mean_length": 44.125, "completions/min_length": 39.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.125, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.8744622468948364, "rewards/meter/std": 0.31945866346359253, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9869142174720764, "rewards/repeat_soft/std": 0.0278518907725811, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.25150617957115173, "rewards/total_composite/mean": 0.7993244528770447, "rewards/total_composite/std": 0.1901601254940033, "reward": 0.7993244528770447, "reward_std": 0.19016008079051971, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14177052676677704, "sampling/sampling_logp_difference/max": 2.6750831604003906, "sampling/importance_sampling_ratio/min": 0.06890109926462173, "sampling/importance_sampling_ratio/mean": 0.9756723642349243, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7793125845491886, "clip_ratio/low_mean": 0.012820512987673283, "clip_ratio/low_min": 0.012820512987673283, "clip_ratio/high_mean": 0.11867795884609222, "clip_ratio/high_max": 0.11867795884609222, "clip_ratio/region_mean": 0.1314984718337655, "reward_total_mean": 0.7993244528770447, "reward_meter_mean": 0.8744622468948364, "reward_meter_std": 0.31945866346359253, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9869142174720764, "reward_repeat_soft_std": 0.0278518907725811, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.25150617957115173, "reward_total_composite_mean": 0.7993244528770447, "reward_total_composite_std": 0.1901601254940033} {"timestamp_utc": "2026-04-12T22:33:36Z", "mode": "train", "global_step": 141, "epoch": 0.014163736815670517, "loss": 0.0086, "grad_norm": 7.010746955871582, "learning_rate": 9.575757575757576e-06, "num_tokens": 272341.0, "completions/mean_length": 198.625, "completions/min_length": 128.0, "completions/max_length": 243.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 198.625, "completions/min_terminated_length": 128.0, "completions/max_terminated_length": 243.0, "rewards/meter/mean": 0.45272260904312134, "rewards/meter/std": 0.27589282393455505, "rewards/count_adherence/mean": 0.9166666269302368, "rewards/count_adherence/std": 0.1259881556034088, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9925938844680786, "rewards/repeat_soft/std": 0.005698754917830229, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.5473595857620239, "rewards/total_composite/std": 0.14514927566051483, "reward": 0.5473595857620239, "reward_std": 0.14514927566051483, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18950483202934265, "sampling/sampling_logp_difference/max": 2.614901542663574, "sampling/importance_sampling_ratio/min": 0.07317499071359634, "sampling/importance_sampling_ratio/mean": 1.0239454507827759, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8596653640270233, "clip_ratio/low_mean": 0.11194692924618721, "clip_ratio/low_min": 0.11194692924618721, "clip_ratio/high_mean": 0.06177238188683987, "clip_ratio/high_max": 0.06177238188683987, "clip_ratio/region_mean": 0.17371931113302708, "reward_total_mean": 0.5473595857620239, "reward_meter_mean": 0.45272260904312134, "reward_meter_std": 0.27589282393455505, "reward_count_adherence_mean": 0.9166666269302368, "reward_count_adherence_std": 0.1259881556034088, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9925938844680786, "reward_repeat_soft_std": 0.005698754917830229, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.5473595857620239, "reward_total_composite_std": 0.14514927566051483} {"timestamp_utc": "2026-04-12T22:33:43Z", "mode": "train", "global_step": 142, "epoch": 0.014264188849824208, "loss": 0.0003, "grad_norm": 12.71095085144043, "learning_rate": 9.572727272727273e-06, "num_tokens": 274053.0, "completions/mean_length": 55.0, "completions/min_length": 45.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.0, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.7612498998641968, "rewards/meter/std": 0.29422006011009216, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 1.0, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.5149999856948853, "rewards/judge_quality/std": 0.16017849743366241, "rewards/total_composite/mean": 0.7470624446868896, "rewards/total_composite/std": 0.13542601466178894, "reward": 0.7470624446868896, "reward_std": 0.13542599976062775, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21361897885799408, "sampling/sampling_logp_difference/max": 1.2875089645385742, "sampling/importance_sampling_ratio/min": 0.2759573459625244, "sampling/importance_sampling_ratio/mean": 1.0175107717514038, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.415143460035324, "clip_ratio/low_mean": 0.06529347132891417, "clip_ratio/low_min": 0.06529347132891417, "clip_ratio/high_mean": 0.11970782186836004, "clip_ratio/high_max": 0.11970782186836004, "clip_ratio/region_mean": 0.1850012931972742, "reward_total_mean": 0.7470624446868896, "reward_meter_mean": 0.7612498998641968, "reward_meter_std": 0.29422006011009216, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 1.0, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.5149999856948853, "reward_judge_quality_std": 0.16017849743366241, "reward_total_composite_mean": 0.7470624446868896, "reward_total_composite_std": 0.13542601466178894} {"timestamp_utc": "2026-04-12T22:33:51Z", "mode": "train", "global_step": 143, "epoch": 0.014364640883977901, "loss": 0.0499, "grad_norm": 7.3252692222595215, "learning_rate": 9.56969696969697e-06, "num_tokens": 277000.0, "completions/mean_length": 181.375, "completions/min_length": 156.0, "completions/max_length": 205.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 181.375, "completions/min_terminated_length": 156.0, "completions/max_terminated_length": 205.0, "rewards/meter/mean": 0.6341123580932617, "rewards/meter/std": 0.3794861435890198, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9955697059631348, "rewards/repeat_soft/std": 0.0051034800708293915, "rewards/judge_quality/mean": 0.3137499690055847, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.4888369143009186, "rewards/total_composite/std": 0.32723328471183777, "reward": 0.4888369143009186, "reward_std": 0.3272332549095154, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21756555140018463, "sampling/sampling_logp_difference/max": 1.8546981811523438, "sampling/importance_sampling_ratio/min": 0.15650016069412231, "sampling/importance_sampling_ratio/mean": 1.031713604927063, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.53104504942894, "clip_ratio/low_mean": 0.060815850272774696, "clip_ratio/low_min": 0.060815850272774696, "clip_ratio/high_mean": 0.13803689181804657, "clip_ratio/high_max": 0.13803689181804657, "clip_ratio/region_mean": 0.19885274209082127, "reward_total_mean": 0.4888369143009186, "reward_meter_mean": 0.6341123580932617, "reward_meter_std": 0.3794861435890198, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9955697059631348, "reward_repeat_soft_std": 0.0051034800708293915, "reward_judge_quality_mean": 0.3137499690055847, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.4888369143009186, "reward_total_composite_std": 0.32723328471183777} {"timestamp_utc": "2026-04-12T22:33:58Z", "mode": "train", "global_step": 144, "epoch": 0.014465092918131592, "loss": 0.0816, "grad_norm": 12.265125274658203, "learning_rate": 9.566666666666668e-06, "num_tokens": 278838.0, "completions/mean_length": 57.75, "completions/min_length": 53.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.75, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.5257716774940491, "rewards/meter/std": 0.47132599353790283, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9956430792808533, "rewards/repeat_soft/std": 0.0068510351702570915, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.19334925711154938, "rewards/total_composite/mean": 0.6267865896224976, "rewards/total_composite/std": 0.20718355476856232, "reward": 0.6267865896224976, "reward_std": 0.20718353986740112, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17005670070648193, "sampling/sampling_logp_difference/max": 1.1533842086791992, "sampling/importance_sampling_ratio/min": 0.3155670166015625, "sampling/importance_sampling_ratio/mean": 1.0278751850128174, "sampling/importance_sampling_ratio/max": 1.8527933359146118, "entropy": 1.7537786066532135, "clip_ratio/low_mean": 0.09149759635329247, "clip_ratio/low_min": 0.09149759635329247, "clip_ratio/high_mean": 0.09468746930360794, "clip_ratio/high_max": 0.09468746930360794, "clip_ratio/region_mean": 0.1861850656569004, "reward_total_mean": 0.6267865896224976, "reward_meter_mean": 0.5257716774940491, "reward_meter_std": 0.47132599353790283, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9956430792808533, "reward_repeat_soft_std": 0.0068510351702570915, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.19334925711154938, "reward_total_composite_mean": 0.6267865896224976, "reward_total_composite_std": 0.20718355476856232} {"timestamp_utc": "2026-04-12T22:34:04Z", "mode": "train", "global_step": 145, "epoch": 0.014565544952285283, "loss": 0.0394, "grad_norm": 15.574225425720215, "learning_rate": 9.563636363636365e-06, "num_tokens": 280473.0, "completions/mean_length": 49.375, "completions/min_length": 32.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.375, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.4346250593662262, "rewards/meter/std": 0.4319024384021759, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9841232299804688, "rewards/repeat_soft/std": 0.01157012116163969, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.263435423374176, "rewards/total_composite/mean": 0.6461185812950134, "rewards/total_composite/std": 0.2213435173034668, "reward": 0.6461185812950134, "reward_std": 0.2213435173034668, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1840735524892807, "sampling/sampling_logp_difference/max": 1.6395959854125977, "sampling/importance_sampling_ratio/min": 0.19405843317508698, "sampling/importance_sampling_ratio/mean": 1.062300443649292, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.101605921983719, "clip_ratio/low_mean": 0.13492169696837664, "clip_ratio/low_min": 0.13492169696837664, "clip_ratio/high_mean": 0.06772094406187534, "clip_ratio/high_max": 0.06772094406187534, "clip_ratio/region_mean": 0.20264264103025198, "reward_total_mean": 0.6461185812950134, "reward_meter_mean": 0.4346250593662262, "reward_meter_std": 0.4319024384021759, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9841232299804688, "reward_repeat_soft_std": 0.01157012116163969, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.263435423374176, "reward_total_composite_mean": 0.6461185812950134, "reward_total_composite_std": 0.2213435173034668} {"timestamp_utc": "2026-04-12T22:34:10Z", "mode": "train", "global_step": 146, "epoch": 0.014665996986438976, "loss": 0.0065, "grad_norm": 14.454992294311523, "learning_rate": 9.56060606060606e-06, "num_tokens": 282280.0, "completions/mean_length": 56.875, "completions/min_length": 41.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.875, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.38828152418136597, "rewards/meter/std": 0.3594237267971039, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9920034408569336, "rewards/repeat_soft/std": 0.016578910872340202, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5499269962310791, "rewards/total_composite/std": 0.16232407093048096, "reward": 0.5499269962310791, "reward_std": 0.16232407093048096, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20530828833580017, "sampling/sampling_logp_difference/max": 1.7603940963745117, "sampling/importance_sampling_ratio/min": 0.17197707295417786, "sampling/importance_sampling_ratio/mean": 1.022547721862793, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.83353291451931, "clip_ratio/low_mean": 0.0978974774479866, "clip_ratio/low_min": 0.0978974774479866, "clip_ratio/high_mean": 0.10358610190451145, "clip_ratio/high_max": 0.10358610190451145, "clip_ratio/region_mean": 0.20148357935249805, "reward_total_mean": 0.5499269962310791, "reward_meter_mean": 0.38828152418136597, "reward_meter_std": 0.3594237267971039, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9920034408569336, "reward_repeat_soft_std": 0.016578910872340202, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5499269962310791, "reward_total_composite_std": 0.16232407093048096} {"timestamp_utc": "2026-04-12T22:34:17Z", "mode": "train", "global_step": 147, "epoch": 0.014766449020592667, "loss": 0.317, "grad_norm": 22.578855514526367, "learning_rate": 9.55757575757576e-06, "num_tokens": 283814.0, "completions/mean_length": 34.75, "completions/min_length": 24.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.75, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.639831006526947, "rewards/meter/std": 0.40390339493751526, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.986702561378479, "rewards/repeat_soft/std": 0.016451282426714897, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.7113442420959473, "rewards/total_composite/std": 0.17334672808647156, "reward": 0.7113442420959473, "reward_std": 0.17334672808647156, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24534133076667786, "sampling/sampling_logp_difference/max": 1.3638458251953125, "sampling/importance_sampling_ratio/min": 0.2556755840778351, "sampling/importance_sampling_ratio/mean": 1.0255770683288574, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8735971450805664, "clip_ratio/low_mean": 0.047843825072050095, "clip_ratio/low_min": 0.047843825072050095, "clip_ratio/high_mean": 0.16567365266382694, "clip_ratio/high_max": 0.16567365266382694, "clip_ratio/region_mean": 0.21351747773587704, "reward_total_mean": 0.7113442420959473, "reward_meter_mean": 0.639831006526947, "reward_meter_std": 0.40390339493751526, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.986702561378479, "reward_repeat_soft_std": 0.016451282426714897, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.7113442420959473, "reward_total_composite_std": 0.17334672808647156} {"timestamp_utc": "2026-04-12T22:34:25Z", "mode": "train", "global_step": 148, "epoch": 0.014866901054746359, "loss": 0.0468, "grad_norm": 8.733606338500977, "learning_rate": 9.554545454545455e-06, "num_tokens": 286450.0, "completions/mean_length": 140.5, "completions/min_length": 83.0, "completions/max_length": 186.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 140.5, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 186.0, "rewards/meter/mean": 0.8556537628173828, "rewards/meter/std": 0.34665367007255554, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9829007983207703, "rewards/repeat_soft/std": 0.016816018149256706, "rewards/judge_quality/mean": 0.39374998211860657, "rewards/judge_quality/std": 0.15638209879398346, "rewards/total_composite/mean": 0.6418274641036987, "rewards/total_composite/std": 0.3184812068939209, "reward": 0.6418274641036987, "reward_std": 0.3184812068939209, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1881764531135559, "sampling/sampling_logp_difference/max": 1.4606990814208984, "sampling/importance_sampling_ratio/min": 0.23207399249076843, "sampling/importance_sampling_ratio/mean": 1.030596137046814, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.944037601351738, "clip_ratio/low_mean": 0.04185550473630428, "clip_ratio/low_min": 0.04185550473630428, "clip_ratio/high_mean": 0.11391049064695835, "clip_ratio/high_max": 0.11391049064695835, "clip_ratio/region_mean": 0.15576599538326263, "reward_total_mean": 0.6418274641036987, "reward_meter_mean": 0.8556537628173828, "reward_meter_std": 0.34665367007255554, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9829007983207703, "reward_repeat_soft_std": 0.016816018149256706, "reward_judge_quality_mean": 0.39374998211860657, "reward_judge_quality_std": 0.15638209879398346, "reward_total_composite_mean": 0.6418274641036987, "reward_total_composite_std": 0.3184812068939209} {"timestamp_utc": "2026-04-12T22:34:33Z", "mode": "train", "global_step": 149, "epoch": 0.01496735308890005, "loss": 0.0005, "grad_norm": 5.917326927185059, "learning_rate": 9.551515151515152e-06, "num_tokens": 289547.0, "completions/mean_length": 196.125, "completions/min_length": 170.0, "completions/max_length": 207.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 196.125, "completions/min_terminated_length": 170.0, "completions/max_terminated_length": 207.0, "rewards/meter/mean": 0.9900409579277039, "rewards/meter/std": 0.01028269249945879, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9094256162643433, "rewards/repeat_soft/std": 0.09267400950193405, "rewards/judge_quality/mean": 0.2212499976158142, "rewards/judge_quality/std": 0.09433034062385559, "rewards/total_composite/mean": 0.657077431678772, "rewards/total_composite/std": 0.2678874731063843, "reward": 0.657077431678772, "reward_std": 0.2678874731063843, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17067110538482666, "sampling/sampling_logp_difference/max": 1.832137107849121, "sampling/importance_sampling_ratio/min": 0.16007111966609955, "sampling/importance_sampling_ratio/mean": 1.0396323204040527, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1036142259836197, "clip_ratio/low_mean": 0.018518518656492233, "clip_ratio/low_min": 0.018518518656492233, "clip_ratio/high_mean": 0.12627116777002811, "clip_ratio/high_max": 0.12627116777002811, "clip_ratio/region_mean": 0.14478968642652035, "reward_total_mean": 0.657077431678772, "reward_meter_mean": 0.9900409579277039, "reward_meter_std": 0.01028269249945879, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9094256162643433, "reward_repeat_soft_std": 0.09267400950193405, "reward_judge_quality_mean": 0.2212499976158142, "reward_judge_quality_std": 0.09433034062385559, "reward_total_composite_mean": 0.657077431678772, "reward_total_composite_std": 0.2678874731063843} {"timestamp_utc": "2026-04-12T22:34:41Z", "mode": "train", "global_step": 150, "epoch": 0.015067805123053743, "loss": 0.0307, "grad_norm": 8.67408275604248, "learning_rate": 9.54848484848485e-06, "num_tokens": 291984.0, "completions/mean_length": 134.625, "completions/min_length": 114.0, "completions/max_length": 164.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 134.625, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 164.0, "rewards/meter/mean": 0.6442027688026428, "rewards/meter/std": 0.23864556849002838, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9940277338027954, "rewards/repeat_soft/std": 0.0051259868778288364, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.641294002532959, "rewards/total_composite/std": 0.1019832044839859, "reward": 0.641294002532959, "reward_std": 0.1019832044839859, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19296503067016602, "sampling/sampling_logp_difference/max": 1.9423902034759521, "sampling/importance_sampling_ratio/min": 0.14336088299751282, "sampling/importance_sampling_ratio/mean": 1.0387064218521118, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8878242373466492, "clip_ratio/low_mean": 0.12727568112313747, "clip_ratio/low_min": 0.12727568112313747, "clip_ratio/high_mean": 0.06800262071192265, "clip_ratio/high_max": 0.06800262071192265, "clip_ratio/region_mean": 0.19527830183506012, "reward_total_mean": 0.641294002532959, "reward_meter_mean": 0.6442027688026428, "reward_meter_std": 0.23864556849002838, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9940277338027954, "reward_repeat_soft_std": 0.0051259868778288364, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.641294002532959, "reward_total_composite_std": 0.1019832044839859} {"timestamp_utc": "2026-04-12T22:35:28Z", "mode": "eval", "global_step": 150, "epoch": 0.015067805123053743, "eval_loss": NaN, "eval_runtime": 47.3495, "eval_samples_per_second": 1.69, "eval_steps_per_second": 0.211, "eval_num_tokens": 291984.0, "eval_completions/mean_length": 89.7, "eval_completions/min_length": 39.4, "eval_completions/max_length": 155.8, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 89.7, "eval_completions/min_terminated_length": 39.4, "eval_completions/max_terminated_length": 155.8, "eval_rewards/meter/mean": 0.579559576511383, "eval_rewards/meter/std": 0.3873100936412811, "eval_rewards/count_adherence/mean": 0.9793750047683716, "eval_rewards/count_adherence/std": 0.046562766283750535, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.07071067690849304, "eval_rewards/repeat_soft/mean": 0.9804000198841095, "eval_rewards/repeat_soft/std": 0.034216074645519255, "eval_rewards/judge_quality/mean": 0.38399999141693114, "eval_rewards/judge_quality/std": 0.13652184307575227, "eval_rewards/total_composite/mean": 0.6085451781749726, "eval_rewards/total_composite/std": 0.1927956983447075, "eval_reward": 0.6085451781749726, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.13093501552939416, "eval_sampling/sampling_logp_difference/max": 1.1314985275268554, "eval_sampling/importance_sampling_ratio/min": 0.3271831452846527, "eval_sampling/importance_sampling_ratio/mean": 1.0350063562393188, "eval_sampling/importance_sampling_ratio/max": 1.5271484375, "eval_entropy": 1.9806707382202149, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6085451781749726, "eval_reward_meter_mean": 0.579559576511383, "eval_reward_meter_std": 0.3873100936412811, "eval_reward_count_adherence_mean": 0.9793750047683716, "eval_reward_count_adherence_std": 0.046562766283750535, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.07071067690849304, "eval_reward_repeat_soft_mean": 0.9804000198841095, "eval_reward_repeat_soft_std": 0.034216074645519255, "eval_reward_judge_quality_mean": 0.38399999141693114, "eval_reward_judge_quality_std": 0.13652184307575227, "eval_reward_total_composite_mean": 0.6085451781749726, "eval_reward_total_composite_std": 0.1927956983447075} {"timestamp_utc": "2026-04-12T22:35:40Z", "mode": "train", "global_step": 151, "epoch": 0.015168257157207434, "loss": -0.0268, "grad_norm": 17.52226448059082, "learning_rate": 9.545454545454547e-06, "num_tokens": 293438.0, "completions/mean_length": 29.75, "completions/min_length": 27.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.75, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.6037045121192932, "rewards/meter/std": 0.4519675076007843, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.960254430770874, "rewards/repeat_soft/std": 0.005280489567667246, "rewards/judge_quality/mean": 0.3687499761581421, "rewards/judge_quality/std": 0.11319231241941452, "rewards/total_composite/mean": 0.6283174753189087, "rewards/total_composite/std": 0.20097284018993378, "reward": 0.6283174753189087, "reward_std": 0.20097284018993378, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2218317985534668, "sampling/sampling_logp_difference/max": 1.4685611724853516, "sampling/importance_sampling_ratio/min": 0.23025654256343842, "sampling/importance_sampling_ratio/mean": 1.0167431831359863, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6521061211824417, "clip_ratio/low_mean": 0.08347017224878073, "clip_ratio/low_min": 0.08347017224878073, "clip_ratio/high_mean": 0.1568104773759842, "clip_ratio/high_max": 0.1568104773759842, "clip_ratio/region_mean": 0.24028064962476492, "reward_total_mean": 0.6283174753189087, "reward_meter_mean": 0.6037045121192932, "reward_meter_std": 0.4519675076007843, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.960254430770874, "reward_repeat_soft_std": 0.005280489567667246, "reward_judge_quality_mean": 0.3687499761581421, "reward_judge_quality_std": 0.11319231241941452, "reward_total_composite_mean": 0.6283174753189087, "reward_total_composite_std": 0.20097284018993378} {"timestamp_utc": "2026-04-12T22:35:47Z", "mode": "train", "global_step": 152, "epoch": 0.015268709191361125, "loss": -0.0616, "grad_norm": 7.388057231903076, "learning_rate": 9.542424242424242e-06, "num_tokens": 295656.0, "completions/mean_length": 114.25, "completions/min_length": 69.0, "completions/max_length": 139.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.25, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.8712646961212158, "rewards/meter/std": 0.28484392166137695, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9923064708709717, "rewards/repeat_soft/std": 0.010224386118352413, "rewards/judge_quality/mean": 0.2887499928474426, "rewards/judge_quality/std": 0.11630470305681229, "rewards/total_composite/mean": 0.7232372164726257, "rewards/total_composite/std": 0.11754972487688065, "reward": 0.7232372164726257, "reward_std": 0.11754971742630005, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19024528563022614, "sampling/sampling_logp_difference/max": 1.115931510925293, "sampling/importance_sampling_ratio/min": 0.32760995626449585, "sampling/importance_sampling_ratio/mean": 1.0422486066818237, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.7057955861091614, "clip_ratio/low_mean": 0.05840101931244135, "clip_ratio/low_min": 0.05840101931244135, "clip_ratio/high_mean": 0.11011046543717384, "clip_ratio/high_max": 0.11011046543717384, "clip_ratio/region_mean": 0.1685114847496152, "reward_total_mean": 0.7232372164726257, "reward_meter_mean": 0.8712646961212158, "reward_meter_std": 0.28484392166137695, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9923064708709717, "reward_repeat_soft_std": 0.010224386118352413, "reward_judge_quality_mean": 0.2887499928474426, "reward_judge_quality_std": 0.11630470305681229, "reward_total_composite_mean": 0.7232372164726257, "reward_total_composite_std": 0.11754972487688065} {"timestamp_utc": "2026-04-12T22:35:58Z", "mode": "train", "global_step": 153, "epoch": 0.015369161225514816, "loss": 0.0206, "grad_norm": 14.170472145080566, "learning_rate": 9.539393939393941e-06, "num_tokens": 297271.0, "completions/mean_length": 55.875, "completions/min_length": 47.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.875, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9694111347198486, "rewards/meter/std": 0.035776134580373764, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9994534254074097, "rewards/repeat_soft/std": 0.001545967417769134, "rewards/judge_quality/mean": 0.44999998807907104, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7175350785255432, "rewards/total_composite/std": 0.29035794734954834, "reward": 0.7175350785255432, "reward_std": 0.29035791754722595, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17416347563266754, "sampling/sampling_logp_difference/max": 1.2880401611328125, "sampling/importance_sampling_ratio/min": 0.2758108079433441, "sampling/importance_sampling_ratio/mean": 1.0363378524780273, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8942160308361053, "clip_ratio/low_mean": 0.02801724150776863, "clip_ratio/low_min": 0.02801724150776863, "clip_ratio/high_mean": 0.13300238270312548, "clip_ratio/high_max": 0.13300238270312548, "clip_ratio/region_mean": 0.1610196242108941, "reward_total_mean": 0.7175350785255432, "reward_meter_mean": 0.9694111347198486, "reward_meter_std": 0.035776134580373764, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9994534254074097, "reward_repeat_soft_std": 0.001545967417769134, "reward_judge_quality_mean": 0.44999998807907104, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7175350785255432, "reward_total_composite_std": 0.29035794734954834} {"timestamp_utc": "2026-04-12T22:36:09Z", "mode": "train", "global_step": 154, "epoch": 0.015469613259668509, "loss": 0.0062, "grad_norm": 14.80764102935791, "learning_rate": 9.536363636363637e-06, "num_tokens": 299067.0, "completions/mean_length": 52.5, "completions/min_length": 36.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.7321591377258301, "rewards/meter/std": 0.37825655937194824, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 1.0, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.3675000071525574, "rewards/judge_quality/std": 0.09808888286352158, "rewards/total_composite/mean": 0.6803466081619263, "rewards/total_composite/std": 0.18387891352176666, "reward": 0.6803466081619263, "reward_std": 0.18387891352176666, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2041114717721939, "sampling/sampling_logp_difference/max": 1.7619848251342773, "sampling/importance_sampling_ratio/min": 0.17170371115207672, "sampling/importance_sampling_ratio/mean": 1.0717298984527588, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.875928968191147, "clip_ratio/low_mean": 0.055980125442147255, "clip_ratio/low_min": 0.055980125442147255, "clip_ratio/high_mean": 0.1288575418293476, "clip_ratio/high_max": 0.1288575418293476, "clip_ratio/region_mean": 0.18483766727149487, "reward_total_mean": 0.6803466081619263, "reward_meter_mean": 0.7321591377258301, "reward_meter_std": 0.37825655937194824, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 1.0, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.3675000071525574, "reward_judge_quality_std": 0.09808888286352158, "reward_total_composite_mean": 0.6803466081619263, "reward_total_composite_std": 0.18387891352176666} {"timestamp_utc": "2026-04-12T22:36:17Z", "mode": "train", "global_step": 155, "epoch": 0.0155700652938222, "loss": 0.0592, "grad_norm": 10.283768653869629, "learning_rate": 9.533333333333334e-06, "num_tokens": 300993.0, "completions/mean_length": 64.75, "completions/min_length": 52.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.75, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.2703203856945038, "rewards/meter/std": 0.4463132321834564, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9945789575576782, "rewards/repeat_soft/std": 0.0068430774845182896, "rewards/judge_quality/mean": 0.4124999940395355, "rewards/judge_quality/std": 0.1685018092393875, "rewards/total_composite/mean": 0.49485206604003906, "rewards/total_composite/std": 0.21574346721172333, "reward": 0.49485206604003906, "reward_std": 0.21574346721172333, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2035830020904541, "sampling/sampling_logp_difference/max": 1.513336181640625, "sampling/importance_sampling_ratio/min": 0.22017420828342438, "sampling/importance_sampling_ratio/mean": 1.0503827333450317, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3192920982837677, "clip_ratio/low_mean": 0.10545289795845747, "clip_ratio/low_min": 0.10545289795845747, "clip_ratio/high_mean": 0.04740618169307709, "clip_ratio/high_max": 0.04740618169307709, "clip_ratio/region_mean": 0.15285907965153456, "reward_total_mean": 0.49485206604003906, "reward_meter_mean": 0.2703203856945038, "reward_meter_std": 0.4463132321834564, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9945789575576782, "reward_repeat_soft_std": 0.0068430774845182896, "reward_judge_quality_mean": 0.4124999940395355, "reward_judge_quality_std": 0.1685018092393875, "reward_total_composite_mean": 0.49485206604003906, "reward_total_composite_std": 0.21574346721172333} {"timestamp_utc": "2026-04-12T22:36:24Z", "mode": "train", "global_step": 156, "epoch": 0.01567051732797589, "loss": -0.0075, "grad_norm": 16.434175491333008, "learning_rate": 9.530303030303031e-06, "num_tokens": 302725.0, "completions/mean_length": 53.5, "completions/min_length": 42.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.5, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.6622544527053833, "rewards/meter/std": 0.35615384578704834, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9921135306358337, "rewards/repeat_soft/std": 0.009797083213925362, "rewards/judge_quality/mean": 0.5012500286102295, "rewards/judge_quality/std": 0.16974246501922607, "rewards/total_composite/mean": 0.6976008415222168, "rewards/total_composite/std": 0.1817840188741684, "reward": 0.6976008415222168, "reward_std": 0.1817840188741684, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16653680801391602, "sampling/sampling_logp_difference/max": 1.999290943145752, "sampling/importance_sampling_ratio/min": 0.13543127477169037, "sampling/importance_sampling_ratio/mean": 1.0143111944198608, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.441454216837883, "clip_ratio/low_mean": 0.07074090465903282, "clip_ratio/low_min": 0.07074090465903282, "clip_ratio/high_mean": 0.08608901780098677, "clip_ratio/high_max": 0.08608901780098677, "clip_ratio/region_mean": 0.1568299224600196, "reward_total_mean": 0.6976008415222168, "reward_meter_mean": 0.6622544527053833, "reward_meter_std": 0.35615384578704834, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9921135306358337, "reward_repeat_soft_std": 0.009797083213925362, "reward_judge_quality_mean": 0.5012500286102295, "reward_judge_quality_std": 0.16974246501922607, "reward_total_composite_mean": 0.6976008415222168, "reward_total_composite_std": 0.1817840188741684} {"timestamp_utc": "2026-04-12T22:36:31Z", "mode": "train", "global_step": 157, "epoch": 0.015770969362129583, "loss": 0.0603, "grad_norm": 20.17525291442871, "learning_rate": 9.527272727272729e-06, "num_tokens": 304186.0, "completions/mean_length": 29.625, "completions/min_length": 25.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.625, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.6452634334564209, "rewards/meter/std": 0.4738078713417053, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9538979530334473, "rewards/repeat_soft/std": 0.024330154061317444, "rewards/judge_quality/mean": 0.2800000011920929, "rewards/judge_quality/std": 0.13125765323638916, "rewards/total_composite/mean": 0.6197583675384521, "rewards/total_composite/std": 0.21760250627994537, "reward": 0.6197583675384521, "reward_std": 0.21760250627994537, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19076097011566162, "sampling/sampling_logp_difference/max": 1.256563663482666, "sampling/importance_sampling_ratio/min": 0.2846304476261139, "sampling/importance_sampling_ratio/mean": 1.0374767780303955, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8032512366771698, "clip_ratio/low_mean": 0.07534630037844181, "clip_ratio/low_min": 0.07534630037844181, "clip_ratio/high_mean": 0.11880760360509157, "clip_ratio/high_max": 0.11880760360509157, "clip_ratio/region_mean": 0.19415390398353338, "reward_total_mean": 0.6197583675384521, "reward_meter_mean": 0.6452634334564209, "reward_meter_std": 0.4738078713417053, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9538979530334473, "reward_repeat_soft_std": 0.024330154061317444, "reward_judge_quality_mean": 0.2800000011920929, "reward_judge_quality_std": 0.13125765323638916, "reward_total_composite_mean": 0.6197583675384521, "reward_total_composite_std": 0.21760250627994537} {"timestamp_utc": "2026-04-12T22:36:38Z", "mode": "train", "global_step": 158, "epoch": 0.015871421396283274, "loss": 0.0186, "grad_norm": 8.423430442810059, "learning_rate": 9.524242424242424e-06, "num_tokens": 305938.0, "completions/mean_length": 71.0, "completions/min_length": 59.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.0, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9905709028244019, "rewards/meter/std": 0.013387288898229599, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9690340161323547, "rewards/repeat_soft/std": 0.03389734774827957, "rewards/judge_quality/mean": 0.36000001430511475, "rewards/judge_quality/std": 0.09165151417255402, "rewards/total_composite/mean": 0.800660252571106, "rewards/total_composite/std": 0.03153316676616669, "reward": 0.800660252571106, "reward_std": 0.03153315186500549, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16707605123519897, "sampling/sampling_logp_difference/max": 1.6637210845947266, "sampling/importance_sampling_ratio/min": 0.1894327700138092, "sampling/importance_sampling_ratio/mean": 1.036516547203064, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6277482360601425, "clip_ratio/low_mean": 0.05309281498193741, "clip_ratio/low_min": 0.05309281498193741, "clip_ratio/high_mean": 0.09371021948754787, "clip_ratio/high_max": 0.09371021948754787, "clip_ratio/region_mean": 0.14680303446948528, "reward_total_mean": 0.800660252571106, "reward_meter_mean": 0.9905709028244019, "reward_meter_std": 0.013387288898229599, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9690340161323547, "reward_repeat_soft_std": 0.03389734774827957, "reward_judge_quality_mean": 0.36000001430511475, "reward_judge_quality_std": 0.09165151417255402, "reward_total_composite_mean": 0.800660252571106, "reward_total_composite_std": 0.03153316676616669} {"timestamp_utc": "2026-04-12T22:36:47Z", "mode": "train", "global_step": 159, "epoch": 0.015971873430436965, "loss": -0.085, "grad_norm": 14.062602996826172, "learning_rate": 9.521212121212121e-06, "num_tokens": 307660.0, "completions/mean_length": 62.25, "completions/min_length": 45.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.25, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9907737374305725, "rewards/meter/std": 0.008397881872951984, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9953635931015015, "rewards/repeat_soft/std": 0.004471874330192804, "rewards/judge_quality/mean": 0.44624999165534973, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8292595744132996, "rewards/total_composite/std": 0.00623986916616559, "reward": 0.8292595744132996, "reward_std": 0.00623986916616559, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17575737833976746, "sampling/sampling_logp_difference/max": 1.4365453720092773, "sampling/importance_sampling_ratio/min": 0.2377476692199707, "sampling/importance_sampling_ratio/mean": 1.0418425798416138, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.062304139137268, "clip_ratio/low_mean": 0.05505712330341339, "clip_ratio/low_min": 0.05505712330341339, "clip_ratio/high_mean": 0.09878544509410858, "clip_ratio/high_max": 0.09878544509410858, "clip_ratio/region_mean": 0.15384256839752197, "reward_total_mean": 0.8292595744132996, "reward_meter_mean": 0.9907737374305725, "reward_meter_std": 0.008397881872951984, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9953635931015015, "reward_repeat_soft_std": 0.004471874330192804, "reward_judge_quality_mean": 0.44624999165534973, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8292595744132996, "reward_total_composite_std": 0.00623986916616559} {"timestamp_utc": "2026-04-12T22:36:54Z", "mode": "train", "global_step": 160, "epoch": 0.01607232546459066, "loss": 0.1267, "grad_norm": 16.92506980895996, "learning_rate": 9.518181818181819e-06, "num_tokens": 309267.0, "completions/mean_length": 43.875, "completions/min_length": 28.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.875, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9226127862930298, "rewards/meter/std": 0.09503024071455002, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9710944890975952, "rewards/repeat_soft/std": 0.05597909912467003, "rewards/judge_quality/mean": 0.6949999928474426, "rewards/judge_quality/std": 0.1908627152442932, "rewards/total_composite/mean": 0.8707852363586426, "rewards/total_composite/std": 0.05670774728059769, "reward": 0.8707852363586426, "reward_std": 0.056707751005887985, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16918140649795532, "sampling/sampling_logp_difference/max": 1.366260051727295, "sampling/importance_sampling_ratio/min": 0.2550590932369232, "sampling/importance_sampling_ratio/mean": 1.0055161714553833, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5125083029270172, "clip_ratio/low_mean": 0.04656171798706055, "clip_ratio/low_min": 0.04656171798706055, "clip_ratio/high_mean": 0.08021787973120809, "clip_ratio/high_max": 0.08021787973120809, "clip_ratio/region_mean": 0.12677959771826863, "reward_total_mean": 0.8707852363586426, "reward_meter_mean": 0.9226127862930298, "reward_meter_std": 0.09503024071455002, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9710944890975952, "reward_repeat_soft_std": 0.05597909912467003, "reward_judge_quality_mean": 0.6949999928474426, "reward_judge_quality_std": 0.1908627152442932, "reward_total_composite_mean": 0.8707852363586426, "reward_total_composite_std": 0.05670774728059769} {"timestamp_utc": "2026-04-12T22:37:02Z", "mode": "train", "global_step": 161, "epoch": 0.01617277749874435, "loss": -0.0626, "grad_norm": 15.525965690612793, "learning_rate": 9.515151515151516e-06, "num_tokens": 310703.0, "completions/mean_length": 33.5, "completions/min_length": 27.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.5, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.8751624822616577, "rewards/meter/std": 0.3393276631832123, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9143266677856445, "rewards/repeat_soft/std": 0.10079991817474365, "rewards/judge_quality/mean": 0.4975000321865082, "rewards/judge_quality/std": 0.2817166745662689, "rewards/total_composite/mean": 0.7845057845115662, "rewards/total_composite/std": 0.20682775974273682, "reward": 0.7845057845115662, "reward_std": 0.20682775974273682, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17265181243419647, "sampling/sampling_logp_difference/max": 1.8043546676635742, "sampling/importance_sampling_ratio/min": 0.16458062827587128, "sampling/importance_sampling_ratio/mean": 1.0136698484420776, "sampling/importance_sampling_ratio/max": 1.8944414854049683, "entropy": 1.8272867798805237, "clip_ratio/low_mean": 0.0351851861923933, "clip_ratio/low_min": 0.0351851861923933, "clip_ratio/high_mean": 0.12637826520949602, "clip_ratio/high_max": 0.12637826520949602, "clip_ratio/region_mean": 0.16156345140188932, "reward_total_mean": 0.7845057845115662, "reward_meter_mean": 0.8751624822616577, "reward_meter_std": 0.3393276631832123, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9143266677856445, "reward_repeat_soft_std": 0.10079991817474365, "reward_judge_quality_mean": 0.4975000321865082, "reward_judge_quality_std": 0.2817166745662689, "reward_total_composite_mean": 0.7845057845115662, "reward_total_composite_std": 0.20682775974273682} {"timestamp_utc": "2026-04-12T22:37:11Z", "mode": "train", "global_step": 162, "epoch": 0.016273229532898042, "loss": 0.0382, "grad_norm": 19.982229232788086, "learning_rate": 9.512121212121213e-06, "num_tokens": 312289.0, "completions/mean_length": 38.25, "completions/min_length": 28.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.25, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.5322328805923462, "rewards/meter/std": 0.3870828151702881, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9937243461608887, "rewards/repeat_soft/std": 0.009791112504899502, "rewards/judge_quality/mean": 0.5349999666213989, "rewards/judge_quality/std": 0.24663449823856354, "rewards/total_composite/mean": 0.6493772268295288, "rewards/total_composite/std": 0.2318180650472641, "reward": 0.6493772268295288, "reward_std": 0.2318180501461029, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22324085235595703, "sampling/sampling_logp_difference/max": 2.7204360961914062, "sampling/importance_sampling_ratio/min": 0.0658460333943367, "sampling/importance_sampling_ratio/mean": 1.00266432762146, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6748038977384567, "clip_ratio/low_mean": 0.09446094930171967, "clip_ratio/low_min": 0.09446094930171967, "clip_ratio/high_mean": 0.0970759429037571, "clip_ratio/high_max": 0.0970759429037571, "clip_ratio/region_mean": 0.19153689220547676, "reward_total_mean": 0.6493772268295288, "reward_meter_mean": 0.5322328805923462, "reward_meter_std": 0.3870828151702881, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9937243461608887, "reward_repeat_soft_std": 0.009791112504899502, "reward_judge_quality_mean": 0.5349999666213989, "reward_judge_quality_std": 0.24663449823856354, "reward_total_composite_mean": 0.6493772268295288, "reward_total_composite_std": 0.2318180650472641} {"timestamp_utc": "2026-04-12T22:37:19Z", "mode": "train", "global_step": 163, "epoch": 0.016373681567051733, "loss": 0.016, "grad_norm": 11.819523811340332, "learning_rate": 9.50909090909091e-06, "num_tokens": 314068.0, "completions/mean_length": 62.375, "completions/min_length": 57.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.375, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9233949780464172, "rewards/meter/std": 0.19471479952335358, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9969461560249329, "rewards/repeat_soft/std": 0.004109133966267109, "rewards/judge_quality/mean": 0.42624998092651367, "rewards/judge_quality/std": 0.14647647738456726, "rewards/total_composite/mean": 0.7930973768234253, "rewards/total_composite/std": 0.11556258052587509, "reward": 0.7930973768234253, "reward_std": 0.1155625730752945, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17588582634925842, "sampling/sampling_logp_difference/max": 1.965134620666504, "sampling/importance_sampling_ratio/min": 0.14013701677322388, "sampling/importance_sampling_ratio/mean": 1.0216656923294067, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6963207721710205, "clip_ratio/low_mean": 0.0425052959471941, "clip_ratio/low_min": 0.0425052959471941, "clip_ratio/high_mean": 0.1385966893285513, "clip_ratio/high_max": 0.1385966893285513, "clip_ratio/region_mean": 0.1811019852757454, "reward_total_mean": 0.7930973768234253, "reward_meter_mean": 0.9233949780464172, "reward_meter_std": 0.19471479952335358, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9969461560249329, "reward_repeat_soft_std": 0.004109133966267109, "reward_judge_quality_mean": 0.42624998092651367, "reward_judge_quality_std": 0.14647647738456726, "reward_total_composite_mean": 0.7930973768234253, "reward_total_composite_std": 0.11556258052587509} {"timestamp_utc": "2026-04-12T22:37:28Z", "mode": "train", "global_step": 164, "epoch": 0.016474133601205424, "loss": 0.0119, "grad_norm": 8.069746971130371, "learning_rate": 9.506060606060606e-06, "num_tokens": 316596.0, "completions/mean_length": 129.0, "completions/min_length": 116.0, "completions/max_length": 139.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 129.0, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.6153008937835693, "rewards/meter/std": 0.3726039528846741, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9899668097496033, "rewards/repeat_soft/std": 0.012333144433796406, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.6226321458816528, "rewards/total_composite/std": 0.19220340251922607, "reward": 0.6226321458816528, "reward_std": 0.19220338761806488, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2005292922258377, "sampling/sampling_logp_difference/max": 1.548935890197754, "sampling/importance_sampling_ratio/min": 0.21247395873069763, "sampling/importance_sampling_ratio/mean": 1.0147041082382202, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.309463620185852, "clip_ratio/low_mean": 0.076393430121243, "clip_ratio/low_min": 0.076393430121243, "clip_ratio/high_mean": 0.09089494496583939, "clip_ratio/high_max": 0.09089494496583939, "clip_ratio/region_mean": 0.16728837508708239, "reward_total_mean": 0.6226321458816528, "reward_meter_mean": 0.6153008937835693, "reward_meter_std": 0.3726039528846741, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9899668097496033, "reward_repeat_soft_std": 0.012333144433796406, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.6226321458816528, "reward_total_composite_std": 0.19220340251922607} {"timestamp_utc": "2026-04-12T22:37:35Z", "mode": "train", "global_step": 165, "epoch": 0.016574585635359115, "loss": 0.0402, "grad_norm": 15.899052619934082, "learning_rate": 9.503030303030303e-06, "num_tokens": 318202.0, "completions/mean_length": 42.75, "completions/min_length": 38.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.75, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.5030492544174194, "rewards/meter/std": 0.417573481798172, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.991955041885376, "rewards/repeat_soft/std": 0.00785687007009983, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.6578176617622375, "rewards/total_composite/std": 0.18415668606758118, "reward": 0.6578176617622375, "reward_std": 0.18415668606758118, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1723388433456421, "sampling/sampling_logp_difference/max": 1.7610225677490234, "sampling/importance_sampling_ratio/min": 0.17186902463436127, "sampling/importance_sampling_ratio/mean": 0.9937219619750977, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7180194333195686, "clip_ratio/low_mean": 0.08987216651439667, "clip_ratio/low_min": 0.08987216651439667, "clip_ratio/high_mean": 0.08828502427786589, "clip_ratio/high_max": 0.08828502427786589, "clip_ratio/region_mean": 0.17815719079226255, "reward_total_mean": 0.6578176617622375, "reward_meter_mean": 0.5030492544174194, "reward_meter_std": 0.417573481798172, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.991955041885376, "reward_repeat_soft_std": 0.00785687007009983, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.6578176617622375, "reward_total_composite_std": 0.18415668606758118} {"timestamp_utc": "2026-04-12T22:37:41Z", "mode": "train", "global_step": 166, "epoch": 0.016675037669512806, "loss": 0.0938, "grad_norm": 13.500005722045898, "learning_rate": 9.5e-06, "num_tokens": 319982.0, "completions/mean_length": 58.5, "completions/min_length": 52.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.5, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.5016564130783081, "rewards/meter/std": 0.40433311462402344, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9944120645523071, "rewards/repeat_soft/std": 0.0066553642973303795, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.5981866121292114, "rewards/total_composite/std": 0.17862769961357117, "reward": 0.5981866121292114, "reward_std": 0.17862768471240997, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1928389072418213, "sampling/sampling_logp_difference/max": 1.3699474334716797, "sampling/importance_sampling_ratio/min": 0.2541202902793884, "sampling/importance_sampling_ratio/mean": 0.9989068508148193, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.914685234427452, "clip_ratio/low_mean": 0.0708869006484747, "clip_ratio/low_min": 0.0708869006484747, "clip_ratio/high_mean": 0.10799220204353333, "clip_ratio/high_max": 0.10799220204353333, "clip_ratio/region_mean": 0.17887910269200802, "reward_total_mean": 0.5981866121292114, "reward_meter_mean": 0.5016564130783081, "reward_meter_std": 0.40433311462402344, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9944120645523071, "reward_repeat_soft_std": 0.0066553642973303795, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.5981866121292114, "reward_total_composite_std": 0.17862769961357117} {"timestamp_utc": "2026-04-12T22:37:49Z", "mode": "train", "global_step": 167, "epoch": 0.016775489703666498, "loss": -0.081, "grad_norm": 11.224985122680664, "learning_rate": 9.496969696969698e-06, "num_tokens": 321684.0, "completions/mean_length": 65.75, "completions/min_length": 45.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.75, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9555763006210327, "rewards/meter/std": 0.06912393122911453, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9968563318252563, "rewards/repeat_soft/std": 0.006129227578639984, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8079450130462646, "rewards/total_composite/std": 0.031966518610715866, "reward": 0.8079450130462646, "reward_std": 0.03196650743484497, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16898435354232788, "sampling/sampling_logp_difference/max": 1.017256736755371, "sampling/importance_sampling_ratio/min": 0.36158549785614014, "sampling/importance_sampling_ratio/mean": 1.0499838590621948, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.367111414670944, "clip_ratio/low_mean": 0.02604166604578495, "clip_ratio/low_min": 0.02604166604578495, "clip_ratio/high_mean": 0.1444350564852357, "clip_ratio/high_max": 0.1444350564852357, "clip_ratio/region_mean": 0.17047672253102064, "reward_total_mean": 0.8079450130462646, "reward_meter_mean": 0.9555763006210327, "reward_meter_std": 0.06912393122911453, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9968563318252563, "reward_repeat_soft_std": 0.006129227578639984, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8079450130462646, "reward_total_composite_std": 0.031966518610715866} {"timestamp_utc": "2026-04-12T22:37:57Z", "mode": "train", "global_step": 168, "epoch": 0.016875941737820192, "loss": 0.0637, "grad_norm": 18.318450927734375, "learning_rate": 9.493939393939395e-06, "num_tokens": 323245.0, "completions/mean_length": 30.125, "completions/min_length": 25.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.125, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.7876026630401611, "rewards/meter/std": 0.35172486305236816, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9564294219017029, "rewards/repeat_soft/std": 0.017170162871479988, "rewards/judge_quality/mean": 0.1875, "rewards/judge_quality/std": 0.0517549142241478, "rewards/total_composite/mean": 0.6563141345977783, "rewards/total_composite/std": 0.15374185144901276, "reward": 0.6563141345977783, "reward_std": 0.15374185144901276, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20536887645721436, "sampling/sampling_logp_difference/max": 1.2958226203918457, "sampling/importance_sampling_ratio/min": 0.2736726403236389, "sampling/importance_sampling_ratio/mean": 1.0338667631149292, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9300547987222672, "clip_ratio/low_mean": 0.04375000111758709, "clip_ratio/low_min": 0.04375000111758709, "clip_ratio/high_mean": 0.1926663313060999, "clip_ratio/high_max": 0.1926663313060999, "clip_ratio/region_mean": 0.23641633242368698, "reward_total_mean": 0.6563141345977783, "reward_meter_mean": 0.7876026630401611, "reward_meter_std": 0.35172486305236816, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9564294219017029, "reward_repeat_soft_std": 0.017170162871479988, "reward_judge_quality_mean": 0.1875, "reward_judge_quality_std": 0.0517549142241478, "reward_total_composite_mean": 0.6563141345977783, "reward_total_composite_std": 0.15374185144901276} {"timestamp_utc": "2026-04-12T22:38:04Z", "mode": "train", "global_step": 169, "epoch": 0.016976393771973883, "loss": 0.0381, "grad_norm": 15.118651390075684, "learning_rate": 9.490909090909092e-06, "num_tokens": 324945.0, "completions/mean_length": 51.5, "completions/min_length": 38.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.5, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.8344469666481018, "rewards/meter/std": 0.3321688175201416, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9956766963005066, "rewards/repeat_soft/std": 0.00431094691157341, "rewards/judge_quality/mean": 0.5437500476837158, "rewards/judge_quality/std": 0.274170845746994, "rewards/total_composite/mean": 0.7881938219070435, "rewards/total_composite/std": 0.18166033923625946, "reward": 0.7881938219070435, "reward_std": 0.18166033923625946, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20305199921131134, "sampling/sampling_logp_difference/max": 1.4444313049316406, "sampling/importance_sampling_ratio/min": 0.23588019609451294, "sampling/importance_sampling_ratio/mean": 1.033147931098938, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.972639039158821, "clip_ratio/low_mean": 0.08102983422577381, "clip_ratio/low_min": 0.08102983422577381, "clip_ratio/high_mean": 0.10669679567217827, "clip_ratio/high_max": 0.10669679567217827, "clip_ratio/region_mean": 0.18772662989795208, "reward_total_mean": 0.7881938219070435, "reward_meter_mean": 0.8344469666481018, "reward_meter_std": 0.3321688175201416, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9956766963005066, "reward_repeat_soft_std": 0.00431094691157341, "reward_judge_quality_mean": 0.5437500476837158, "reward_judge_quality_std": 0.274170845746994, "reward_total_composite_mean": 0.7881938219070435, "reward_total_composite_std": 0.18166033923625946} {"timestamp_utc": "2026-04-12T22:38:10Z", "mode": "train", "global_step": 170, "epoch": 0.017076845806127575, "loss": -0.1136, "grad_norm": 11.33247184753418, "learning_rate": 9.487878787878788e-06, "num_tokens": 326605.0, "completions/mean_length": 60.5, "completions/min_length": 32.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.5, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.6450315713882446, "rewards/meter/std": 0.43944719433784485, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.994788408279419, "rewards/repeat_soft/std": 0.006934575270861387, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.21084439754486084, "rewards/total_composite/mean": 0.691618025302887, "rewards/total_composite/std": 0.21619349718093872, "reward": 0.691618025302887, "reward_std": 0.21619349718093872, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19176585972309113, "sampling/sampling_logp_difference/max": 1.2366342544555664, "sampling/importance_sampling_ratio/min": 0.29035985469818115, "sampling/importance_sampling_ratio/mean": 1.0253686904907227, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2508623600006104, "clip_ratio/low_mean": 0.06805959902703762, "clip_ratio/low_min": 0.06805959902703762, "clip_ratio/high_mean": 0.10586905479431152, "clip_ratio/high_max": 0.10586905479431152, "clip_ratio/region_mean": 0.17392865382134914, "reward_total_mean": 0.691618025302887, "reward_meter_mean": 0.6450315713882446, "reward_meter_std": 0.43944719433784485, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.994788408279419, "reward_repeat_soft_std": 0.006934575270861387, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.21084439754486084, "reward_total_composite_mean": 0.691618025302887, "reward_total_composite_std": 0.21619349718093872} {"timestamp_utc": "2026-04-12T22:38:17Z", "mode": "train", "global_step": 171, "epoch": 0.017177297840281266, "loss": 0.0733, "grad_norm": 12.215380668640137, "learning_rate": 9.484848484848485e-06, "num_tokens": 328431.0, "completions/mean_length": 62.25, "completions/min_length": 56.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.25, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.7637922167778015, "rewards/meter/std": 0.3509538173675537, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9932276606559753, "rewards/repeat_soft/std": 0.01168556697666645, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7212792634963989, "rewards/total_composite/std": 0.15578024089336395, "reward": 0.7212792634963989, "reward_std": 0.15578024089336395, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18471170961856842, "sampling/sampling_logp_difference/max": 2.0597496032714844, "sampling/importance_sampling_ratio/min": 0.12748588621616364, "sampling/importance_sampling_ratio/mean": 1.0541846752166748, "sampling/importance_sampling_ratio/max": 1.9332194328308105, "entropy": 1.9895541220903397, "clip_ratio/low_mean": 0.07853026315569878, "clip_ratio/low_min": 0.07853026315569878, "clip_ratio/high_mean": 0.11501851491630077, "clip_ratio/high_max": 0.11501851491630077, "clip_ratio/region_mean": 0.19354877807199955, "reward_total_mean": 0.7212792634963989, "reward_meter_mean": 0.7637922167778015, "reward_meter_std": 0.3509538173675537, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9932276606559753, "reward_repeat_soft_std": 0.01168556697666645, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7212792634963989, "reward_total_composite_std": 0.15578024089336395} {"timestamp_utc": "2026-04-12T22:38:27Z", "mode": "train", "global_step": 172, "epoch": 0.017277749874434957, "loss": 0.0415, "grad_norm": 13.837084770202637, "learning_rate": 9.481818181818182e-06, "num_tokens": 330178.0, "completions/mean_length": 55.375, "completions/min_length": 50.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.375, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.805716872215271, "rewards/meter/std": 0.3166608512401581, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9853614568710327, "rewards/repeat_soft/std": 0.01756989397108555, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.7206087112426758, "rewards/total_composite/std": 0.1358690857887268, "reward": 0.7206087112426758, "reward_std": 0.1358690708875656, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13992270827293396, "sampling/sampling_logp_difference/max": 1.246485710144043, "sampling/importance_sampling_ratio/min": 0.2875134348869324, "sampling/importance_sampling_ratio/mean": 0.9956533908843994, "sampling/importance_sampling_ratio/max": 1.9486217498779297, "entropy": 1.2204567715525627, "clip_ratio/low_mean": 0.012500000186264515, "clip_ratio/low_min": 0.012500000186264515, "clip_ratio/high_mean": 0.1423769574612379, "clip_ratio/high_max": 0.1423769574612379, "clip_ratio/region_mean": 0.15487695764750242, "reward_total_mean": 0.7206087112426758, "reward_meter_mean": 0.805716872215271, "reward_meter_std": 0.3166608512401581, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9853614568710327, "reward_repeat_soft_std": 0.01756989397108555, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.7206087112426758, "reward_total_composite_std": 0.1358690857887268} {"timestamp_utc": "2026-04-12T22:38:34Z", "mode": "train", "global_step": 173, "epoch": 0.017378201908588648, "loss": 0.0253, "grad_norm": 17.61992645263672, "learning_rate": 9.47878787878788e-06, "num_tokens": 331758.0, "completions/mean_length": 33.5, "completions/min_length": 26.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.5, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.46493038535118103, "rewards/meter/std": 0.39277565479278564, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.6650000214576721, "rewards/judge_quality/std": 0.3616628050804138, "rewards/total_composite/mean": 0.6549686789512634, "rewards/total_composite/std": 0.2746511399745941, "reward": 0.6549686789512634, "reward_std": 0.2746511399745941, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22067639231681824, "sampling/sampling_logp_difference/max": 1.2074518203735352, "sampling/importance_sampling_ratio/min": 0.29895809292793274, "sampling/importance_sampling_ratio/mean": 0.9965046644210815, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8067358434200287, "clip_ratio/low_mean": 0.09746044129133224, "clip_ratio/low_min": 0.09746044129133224, "clip_ratio/high_mean": 0.15546264499425888, "clip_ratio/high_max": 0.15546264499425888, "clip_ratio/region_mean": 0.2529230862855911, "reward_total_mean": 0.6549686789512634, "reward_meter_mean": 0.46493038535118103, "reward_meter_std": 0.39277565479278564, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.6650000214576721, "reward_judge_quality_std": 0.3616628050804138, "reward_total_composite_mean": 0.6549686789512634, "reward_total_composite_std": 0.2746511399745941} {"timestamp_utc": "2026-04-12T22:38:41Z", "mode": "train", "global_step": 174, "epoch": 0.01747865394274234, "loss": 0.0398, "grad_norm": 11.458572387695312, "learning_rate": 9.475757575757577e-06, "num_tokens": 333544.0, "completions/mean_length": 66.25, "completions/min_length": 57.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.25, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.7040383815765381, "rewards/meter/std": 0.4332011938095093, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9906487464904785, "rewards/repeat_soft/std": 0.0084311468526721, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.5697088241577148, "rewards/total_composite/std": 0.3074597716331482, "reward": 0.5697088241577148, "reward_std": 0.3074597418308258, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21156510710716248, "sampling/sampling_logp_difference/max": 1.6912975311279297, "sampling/importance_sampling_ratio/min": 0.18428026139736176, "sampling/importance_sampling_ratio/mean": 1.035260558128357, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.488626852631569, "clip_ratio/low_mean": 0.062451123259961605, "clip_ratio/low_min": 0.062451123259961605, "clip_ratio/high_mean": 0.11698358226567507, "clip_ratio/high_max": 0.11698358226567507, "clip_ratio/region_mean": 0.17943470552563667, "reward_total_mean": 0.5697088241577148, "reward_meter_mean": 0.7040383815765381, "reward_meter_std": 0.4332011938095093, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9906487464904785, "reward_repeat_soft_std": 0.0084311468526721, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.5697088241577148, "reward_total_composite_std": 0.3074597716331482} {"timestamp_utc": "2026-04-12T22:38:48Z", "mode": "train", "global_step": 175, "epoch": 0.01757910597689603, "loss": 0.011, "grad_norm": 12.192331314086914, "learning_rate": 9.472727272727274e-06, "num_tokens": 335228.0, "completions/mean_length": 55.5, "completions/min_length": 39.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.5, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.6966537833213806, "rewards/meter/std": 0.3916904330253601, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9995148181915283, "rewards/repeat_soft/std": 0.001061148359440267, "rewards/judge_quality/mean": 0.5612499713897705, "rewards/judge_quality/std": 0.25614938139915466, "rewards/total_composite/mean": 0.7318206429481506, "rewards/total_composite/std": 0.21793721616268158, "reward": 0.7318206429481506, "reward_std": 0.2179371863603592, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22025910019874573, "sampling/sampling_logp_difference/max": 1.8662443161010742, "sampling/importance_sampling_ratio/min": 0.15470360219478607, "sampling/importance_sampling_ratio/mean": 1.003103256225586, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9368027225136757, "clip_ratio/low_mean": 0.06872858293354511, "clip_ratio/low_min": 0.06872858293354511, "clip_ratio/high_mean": 0.1247611390426755, "clip_ratio/high_max": 0.1247611390426755, "clip_ratio/region_mean": 0.1934897219762206, "reward_total_mean": 0.7318206429481506, "reward_meter_mean": 0.6966537833213806, "reward_meter_std": 0.3916904330253601, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9995148181915283, "reward_repeat_soft_std": 0.001061148359440267, "reward_judge_quality_mean": 0.5612499713897705, "reward_judge_quality_std": 0.25614938139915466, "reward_total_composite_mean": 0.7318206429481506, "reward_total_composite_std": 0.21793721616268158} {"timestamp_utc": "2026-04-12T22:38:55Z", "mode": "train", "global_step": 176, "epoch": 0.017679558011049725, "loss": 0.0106, "grad_norm": 17.778940200805664, "learning_rate": 9.469696969696971e-06, "num_tokens": 336807.0, "completions/mean_length": 39.375, "completions/min_length": 34.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.375, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.636339008808136, "rewards/meter/std": 0.36255964636802673, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9961650967597961, "rewards/repeat_soft/std": 0.004527193494141102, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.6502262353897095, "rewards/total_composite/std": 0.27786290645599365, "reward": 0.6502262353897095, "reward_std": 0.27786290645599365, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2240709811449051, "sampling/sampling_logp_difference/max": 2.3256683349609375, "sampling/importance_sampling_ratio/min": 0.09771811217069626, "sampling/importance_sampling_ratio/mean": 1.0258606672286987, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.152998335659504, "clip_ratio/low_mean": 0.049544818699359894, "clip_ratio/low_min": 0.049544818699359894, "clip_ratio/high_mean": 0.14569118106737733, "clip_ratio/high_max": 0.14569118106737733, "clip_ratio/region_mean": 0.19523599976673722, "reward_total_mean": 0.6502262353897095, "reward_meter_mean": 0.636339008808136, "reward_meter_std": 0.36255964636802673, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9961650967597961, "reward_repeat_soft_std": 0.004527193494141102, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.6502262353897095, "reward_total_composite_std": 0.27786290645599365} {"timestamp_utc": "2026-04-12T22:39:01Z", "mode": "train", "global_step": 177, "epoch": 0.017780010045203416, "loss": 0.1156, "grad_norm": 13.959003448486328, "learning_rate": 9.466666666666667e-06, "num_tokens": 338529.0, "completions/mean_length": 64.25, "completions/min_length": 56.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.25, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.6710958480834961, "rewards/meter/std": 0.39831477403640747, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9918336868286133, "rewards/repeat_soft/std": 0.021483546122908592, "rewards/judge_quality/mean": 0.44875001907348633, "rewards/judge_quality/std": 0.21256513893604279, "rewards/total_composite/mean": 0.6858015060424805, "rewards/total_composite/std": 0.2134513258934021, "reward": 0.6858015060424805, "reward_std": 0.2134513109922409, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1920711249113083, "sampling/sampling_logp_difference/max": 1.3209571838378906, "sampling/importance_sampling_ratio/min": 0.26687970757484436, "sampling/importance_sampling_ratio/mean": 1.0333881378173828, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0870207250118256, "clip_ratio/low_mean": 0.048370727337896824, "clip_ratio/low_min": 0.048370727337896824, "clip_ratio/high_mean": 0.10638105683028698, "clip_ratio/high_max": 0.10638105683028698, "clip_ratio/region_mean": 0.1547517841681838, "reward_total_mean": 0.6858015060424805, "reward_meter_mean": 0.6710958480834961, "reward_meter_std": 0.39831477403640747, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9918336868286133, "reward_repeat_soft_std": 0.021483546122908592, "reward_judge_quality_mean": 0.44875001907348633, "reward_judge_quality_std": 0.21256513893604279, "reward_total_composite_mean": 0.6858015060424805, "reward_total_composite_std": 0.2134513258934021} {"timestamp_utc": "2026-04-12T22:39:11Z", "mode": "train", "global_step": 178, "epoch": 0.017880462079357107, "loss": 0.0151, "grad_norm": 6.557941913604736, "learning_rate": 9.463636363636364e-06, "num_tokens": 341373.0, "completions/mean_length": 157.5, "completions/min_length": 128.0, "completions/max_length": 186.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 157.5, "completions/min_terminated_length": 128.0, "completions/max_terminated_length": 186.0, "rewards/meter/mean": 0.8198948502540588, "rewards/meter/std": 0.29119426012039185, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9922699332237244, "rewards/repeat_soft/std": 0.010870981961488724, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.7213046550750732, "rewards/total_composite/std": 0.15226486325263977, "reward": 0.7213046550750732, "reward_std": 0.15226484835147858, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18987901508808136, "sampling/sampling_logp_difference/max": 1.669508934020996, "sampling/importance_sampling_ratio/min": 0.18833953142166138, "sampling/importance_sampling_ratio/mean": 1.0420938730239868, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5444493293762207, "clip_ratio/low_mean": 0.03958333469927311, "clip_ratio/low_min": 0.03958333469927311, "clip_ratio/high_mean": 0.1338949054479599, "clip_ratio/high_max": 0.1338949054479599, "clip_ratio/region_mean": 0.173478240147233, "reward_total_mean": 0.7213046550750732, "reward_meter_mean": 0.8198948502540588, "reward_meter_std": 0.29119426012039185, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9922699332237244, "reward_repeat_soft_std": 0.010870981961488724, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.7213046550750732, "reward_total_composite_std": 0.15226486325263977} {"timestamp_utc": "2026-04-12T22:39:18Z", "mode": "train", "global_step": 179, "epoch": 0.0179809141135108, "loss": 0.0451, "grad_norm": 7.193534851074219, "learning_rate": 9.460606060606061e-06, "num_tokens": 343837.0, "completions/mean_length": 132.0, "completions/min_length": 120.0, "completions/max_length": 147.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.0, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 147.0, "rewards/meter/mean": 0.6529181599617004, "rewards/meter/std": 0.33133381605148315, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9972844123840332, "rewards/repeat_soft/std": 0.00252185994759202, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.6744166016578674, "rewards/total_composite/std": 0.14660386741161346, "reward": 0.6744166016578674, "reward_std": 0.14660383760929108, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18584877252578735, "sampling/sampling_logp_difference/max": 1.4600629806518555, "sampling/importance_sampling_ratio/min": 0.23222164809703827, "sampling/importance_sampling_ratio/mean": 1.0315757989883423, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.474087432026863, "clip_ratio/low_mean": 0.06770833395421505, "clip_ratio/low_min": 0.06770833395421505, "clip_ratio/high_mean": 0.11158846318721771, "clip_ratio/high_max": 0.11158846318721771, "clip_ratio/region_mean": 0.17929679714143276, "reward_total_mean": 0.6744166016578674, "reward_meter_mean": 0.6529181599617004, "reward_meter_std": 0.33133381605148315, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9972844123840332, "reward_repeat_soft_std": 0.00252185994759202, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.6744166016578674, "reward_total_composite_std": 0.14660386741161346} {"timestamp_utc": "2026-04-12T22:39:26Z", "mode": "train", "global_step": 180, "epoch": 0.01808136614766449, "loss": -0.0355, "grad_norm": 10.06341552734375, "learning_rate": 9.457575757575759e-06, "num_tokens": 346293.0, "completions/mean_length": 109.0, "completions/min_length": 88.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.0, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.4708794951438904, "rewards/meter/std": 0.40037837624549866, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9943361282348633, "rewards/repeat_soft/std": 0.00587871577590704, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.12351980060338974, "rewards/total_composite/mean": 0.45544177293777466, "rewards/total_composite/std": 0.25155875086784363, "reward": 0.45544177293777466, "reward_std": 0.25155875086784363, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20670326054096222, "sampling/sampling_logp_difference/max": 1.546766757965088, "sampling/importance_sampling_ratio/min": 0.21293532848358154, "sampling/importance_sampling_ratio/mean": 1.0337767601013184, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5227967500686646, "clip_ratio/low_mean": 0.1189667284488678, "clip_ratio/low_min": 0.1189667284488678, "clip_ratio/high_mean": 0.0869832169264555, "clip_ratio/high_max": 0.0869832169264555, "clip_ratio/region_mean": 0.2059499453753233, "reward_total_mean": 0.45544177293777466, "reward_meter_mean": 0.4708794951438904, "reward_meter_std": 0.40037837624549866, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9943361282348633, "reward_repeat_soft_std": 0.00587871577590704, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.12351980060338974, "reward_total_composite_mean": 0.45544177293777466, "reward_total_composite_std": 0.25155875086784363} {"timestamp_utc": "2026-04-12T22:39:32Z", "mode": "train", "global_step": 181, "epoch": 0.01818181818181818, "loss": 0.0199, "grad_norm": 11.772783279418945, "learning_rate": 9.454545454545456e-06, "num_tokens": 348100.0, "completions/mean_length": 65.875, "completions/min_length": 59.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.875, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.7274728417396545, "rewards/meter/std": 0.3568710684776306, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9997270107269287, "rewards/repeat_soft/std": 0.0007720249122940004, "rewards/judge_quality/mean": 0.5212500095367432, "rewards/judge_quality/std": 0.25837060809135437, "rewards/total_composite/mean": 0.733710527420044, "rewards/total_composite/std": 0.15925608575344086, "reward": 0.733710527420044, "reward_std": 0.15925607085227966, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16465792059898376, "sampling/sampling_logp_difference/max": 0.9982872009277344, "sampling/importance_sampling_ratio/min": 0.36851009726524353, "sampling/importance_sampling_ratio/mean": 1.0380871295928955, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9717163443565369, "clip_ratio/low_mean": 0.04475616663694382, "clip_ratio/low_min": 0.04475616663694382, "clip_ratio/high_mean": 0.09800355415791273, "clip_ratio/high_max": 0.09800355415791273, "clip_ratio/region_mean": 0.14275972079485655, "reward_total_mean": 0.733710527420044, "reward_meter_mean": 0.7274728417396545, "reward_meter_std": 0.3568710684776306, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9997270107269287, "reward_repeat_soft_std": 0.0007720249122940004, "reward_judge_quality_mean": 0.5212500095367432, "reward_judge_quality_std": 0.25837060809135437, "reward_total_composite_mean": 0.733710527420044, "reward_total_composite_std": 0.15925608575344086} {"timestamp_utc": "2026-04-12T22:39:38Z", "mode": "train", "global_step": 182, "epoch": 0.018282270215971872, "loss": 0.1234, "grad_norm": 28.719383239746094, "learning_rate": 9.451515151515153e-06, "num_tokens": 349735.0, "completions/mean_length": 40.375, "completions/min_length": 26.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.375, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.6474294662475586, "rewards/meter/std": 0.36743947863578796, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9840507507324219, "rewards/repeat_soft/std": 0.022401679307222366, "rewards/judge_quality/mean": 0.5149999856948853, "rewards/judge_quality/std": 0.2676885426044464, "rewards/total_composite/mean": 0.6942483186721802, "rewards/total_composite/std": 0.19757404923439026, "reward": 0.6942483186721802, "reward_std": 0.19757404923439026, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2344406545162201, "sampling/sampling_logp_difference/max": 2.6010918617248535, "sampling/importance_sampling_ratio/min": 0.07419252395629883, "sampling/importance_sampling_ratio/mean": 0.9845522046089172, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.226590447127819, "clip_ratio/low_mean": 0.03371212258934975, "clip_ratio/low_min": 0.03371212258934975, "clip_ratio/high_mean": 0.17729218304157257, "clip_ratio/high_max": 0.17729218304157257, "clip_ratio/region_mean": 0.21100430563092232, "reward_total_mean": 0.6942483186721802, "reward_meter_mean": 0.6474294662475586, "reward_meter_std": 0.36743947863578796, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9840507507324219, "reward_repeat_soft_std": 0.022401679307222366, "reward_judge_quality_mean": 0.5149999856948853, "reward_judge_quality_std": 0.2676885426044464, "reward_total_composite_mean": 0.6942483186721802, "reward_total_composite_std": 0.19757404923439026} {"timestamp_utc": "2026-04-12T22:39:45Z", "mode": "train", "global_step": 183, "epoch": 0.018382722250125567, "loss": 0.1097, "grad_norm": 15.859298706054688, "learning_rate": 9.448484848484849e-06, "num_tokens": 351498.0, "completions/mean_length": 47.375, "completions/min_length": 40.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.375, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.2945380210876465, "rewards/meter/std": 0.2874630093574524, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9951648116111755, "rewards/repeat_soft/std": 0.005753299221396446, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.5313085913658142, "rewards/total_composite/std": 0.13464023172855377, "reward": 0.5313085913658142, "reward_std": 0.13464021682739258, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22787530720233917, "sampling/sampling_logp_difference/max": 2.0343985557556152, "sampling/importance_sampling_ratio/min": 0.1307591050863266, "sampling/importance_sampling_ratio/mean": 1.0067754983901978, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.418655201792717, "clip_ratio/low_mean": 0.12804551888257265, "clip_ratio/low_min": 0.12804551888257265, "clip_ratio/high_mean": 0.09737318754196167, "clip_ratio/high_max": 0.09737318754196167, "clip_ratio/region_mean": 0.22541870642453432, "reward_total_mean": 0.5313085913658142, "reward_meter_mean": 0.2945380210876465, "reward_meter_std": 0.2874630093574524, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9951648116111755, "reward_repeat_soft_std": 0.005753299221396446, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.5313085913658142, "reward_total_composite_std": 0.13464023172855377} {"timestamp_utc": "2026-04-12T22:39:52Z", "mode": "train", "global_step": 184, "epoch": 0.018483174284279258, "loss": 0.0429, "grad_norm": 10.949633598327637, "learning_rate": 9.445454545454546e-06, "num_tokens": 353266.0, "completions/mean_length": 65.0, "completions/min_length": 59.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.6457832455635071, "rewards/meter/std": 0.40137815475463867, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9987822771072388, "rewards/repeat_soft/std": 0.001760914921760559, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.6541056632995605, "rewards/total_composite/std": 0.18264658749103546, "reward": 0.6541056632995605, "reward_std": 0.18264657258987427, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17986838519573212, "sampling/sampling_logp_difference/max": 1.1473078727722168, "sampling/importance_sampling_ratio/min": 0.3174903392791748, "sampling/importance_sampling_ratio/mean": 1.0141305923461914, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1695086508989334, "clip_ratio/low_mean": 0.055078125558793545, "clip_ratio/low_min": 0.055078125558793545, "clip_ratio/high_mean": 0.12554166465997696, "clip_ratio/high_max": 0.12554166465997696, "clip_ratio/region_mean": 0.1806197902187705, "reward_total_mean": 0.6541056632995605, "reward_meter_mean": 0.6457832455635071, "reward_meter_std": 0.40137815475463867, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9987822771072388, "reward_repeat_soft_std": 0.001760914921760559, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.6541056632995605, "reward_total_composite_std": 0.18264658749103546} {"timestamp_utc": "2026-04-12T22:39:58Z", "mode": "train", "global_step": 185, "epoch": 0.01858362631843295, "loss": 0.0067, "grad_norm": 18.104721069335938, "learning_rate": 9.442424242424243e-06, "num_tokens": 354728.0, "completions/mean_length": 30.75, "completions/min_length": 26.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.75, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.6661310791969299, "rewards/meter/std": 0.34061819314956665, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9516666531562805, "rewards/repeat_soft/std": 0.024608036503195763, "rewards/judge_quality/mean": 0.35999998450279236, "rewards/judge_quality/std": 0.09165150672197342, "rewards/total_composite/mean": 0.6529256105422974, "rewards/total_composite/std": 0.15953470766544342, "reward": 0.6529256105422974, "reward_std": 0.15953470766544342, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17186814546585083, "sampling/sampling_logp_difference/max": 1.139305591583252, "sampling/importance_sampling_ratio/min": 0.3200412094593048, "sampling/importance_sampling_ratio/mean": 1.0297634601593018, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8394776284694672, "clip_ratio/low_mean": 0.04160380829125643, "clip_ratio/low_min": 0.04160380829125643, "clip_ratio/high_mean": 0.10947296489030123, "clip_ratio/high_max": 0.10947296489030123, "clip_ratio/region_mean": 0.15107677318155766, "reward_total_mean": 0.6529256105422974, "reward_meter_mean": 0.6661310791969299, "reward_meter_std": 0.34061819314956665, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9516666531562805, "reward_repeat_soft_std": 0.024608036503195763, "reward_judge_quality_mean": 0.35999998450279236, "reward_judge_quality_std": 0.09165150672197342, "reward_total_composite_mean": 0.6529256105422974, "reward_total_composite_std": 0.15953470766544342} {"timestamp_utc": "2026-04-12T22:40:06Z", "mode": "train", "global_step": 186, "epoch": 0.01868407835258664, "loss": 0.0512, "grad_norm": 11.477437019348145, "learning_rate": 9.43939393939394e-06, "num_tokens": 356878.0, "completions/mean_length": 85.75, "completions/min_length": 81.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.75, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.33967459201812744, "rewards/meter/std": 0.3127252459526062, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9931557774543762, "rewards/repeat_soft/std": 0.00534589309245348, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.5217941403388977, "rewards/total_composite/std": 0.14443467557430267, "reward": 0.5217941403388977, "reward_std": 0.14443467557430267, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21810176968574524, "sampling/sampling_logp_difference/max": 4.856741428375244, "sampling/importance_sampling_ratio/min": 0.007775780279189348, "sampling/importance_sampling_ratio/mean": 1.0244901180267334, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3468519300222397, "clip_ratio/low_mean": 0.11296561546623707, "clip_ratio/low_min": 0.11296561546623707, "clip_ratio/high_mean": 0.07902637496590614, "clip_ratio/high_max": 0.07902637496590614, "clip_ratio/region_mean": 0.1919919904321432, "reward_total_mean": 0.5217941403388977, "reward_meter_mean": 0.33967459201812744, "reward_meter_std": 0.3127252459526062, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9931557774543762, "reward_repeat_soft_std": 0.00534589309245348, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.5217941403388977, "reward_total_composite_std": 0.14443467557430267} {"timestamp_utc": "2026-04-12T22:40:15Z", "mode": "train", "global_step": 187, "epoch": 0.01878453038674033, "loss": -0.0318, "grad_norm": 5.909311294555664, "learning_rate": 9.436363636363636e-06, "num_tokens": 360033.0, "completions/mean_length": 196.375, "completions/min_length": 166.0, "completions/max_length": 226.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 196.375, "completions/min_terminated_length": 166.0, "completions/max_terminated_length": 226.0, "rewards/meter/mean": 0.7298188805580139, "rewards/meter/std": 0.3511499762535095, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9943335056304932, "rewards/repeat_soft/std": 0.00467671500518918, "rewards/judge_quality/mean": 0.29249998927116394, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.5305139422416687, "rewards/total_composite/std": 0.3508707284927368, "reward": 0.5305139422416687, "reward_std": 0.35087069869041443, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1829221248626709, "sampling/sampling_logp_difference/max": 1.5122761726379395, "sampling/importance_sampling_ratio/min": 0.22040772438049316, "sampling/importance_sampling_ratio/mean": 1.0364093780517578, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3696429282426834, "clip_ratio/low_mean": 0.05853912979364395, "clip_ratio/low_min": 0.05853912979364395, "clip_ratio/high_mean": 0.10988106019794941, "clip_ratio/high_max": 0.10988106019794941, "clip_ratio/region_mean": 0.16842018999159336, "reward_total_mean": 0.5305139422416687, "reward_meter_mean": 0.7298188805580139, "reward_meter_std": 0.3511499762535095, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9943335056304932, "reward_repeat_soft_std": 0.00467671500518918, "reward_judge_quality_mean": 0.29249998927116394, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.5305139422416687, "reward_total_composite_std": 0.3508707284927368} {"timestamp_utc": "2026-04-12T22:40:22Z", "mode": "train", "global_step": 188, "epoch": 0.018884982420894023, "loss": -0.0028, "grad_norm": 10.292993545532227, "learning_rate": 9.433333333333335e-06, "num_tokens": 361852.0, "completions/mean_length": 70.375, "completions/min_length": 66.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.375, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9887566566467285, "rewards/meter/std": 0.022077880799770355, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9968456029891968, "rewards/repeat_soft/std": 0.0058901249431073666, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.8176250457763672, "rewards/total_composite/std": 0.02950715832412243, "reward": 0.8176250457763672, "reward_std": 0.029507163912057877, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1563209444284439, "sampling/sampling_logp_difference/max": 1.2766375541687012, "sampling/importance_sampling_ratio/min": 0.2789737284183502, "sampling/importance_sampling_ratio/mean": 1.0275911092758179, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0365912318229675, "clip_ratio/low_mean": 0.016791045665740967, "clip_ratio/low_min": 0.016791045665740967, "clip_ratio/high_mean": 0.15039433352649212, "clip_ratio/high_max": 0.15039433352649212, "clip_ratio/region_mean": 0.16718537919223309, "reward_total_mean": 0.8176250457763672, "reward_meter_mean": 0.9887566566467285, "reward_meter_std": 0.022077880799770355, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9968456029891968, "reward_repeat_soft_std": 0.0058901249431073666, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.8176250457763672, "reward_total_composite_std": 0.02950715832412243} {"timestamp_utc": "2026-04-12T22:40:29Z", "mode": "train", "global_step": 189, "epoch": 0.018985434455047714, "loss": 0.1136, "grad_norm": 26.591156005859375, "learning_rate": 9.43030303030303e-06, "num_tokens": 363424.0, "completions/mean_length": 40.5, "completions/min_length": 31.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.5, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.3919089734554291, "rewards/meter/std": 0.3436703681945801, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9904083013534546, "rewards/repeat_soft/std": 0.012789538130164146, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.663899838924408, "rewards/total_composite/std": 0.15209518373012543, "reward": 0.663899838924408, "reward_std": 0.15209518373012543, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22401906549930573, "sampling/sampling_logp_difference/max": 1.9713678359985352, "sampling/importance_sampling_ratio/min": 0.1392662227153778, "sampling/importance_sampling_ratio/mean": 1.0248298645019531, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4507956057786942, "clip_ratio/low_mean": 0.11950287409126759, "clip_ratio/low_min": 0.11950287409126759, "clip_ratio/high_mean": 0.10845978185534477, "clip_ratio/high_max": 0.10845978185534477, "clip_ratio/region_mean": 0.22796265594661236, "reward_total_mean": 0.663899838924408, "reward_meter_mean": 0.3919089734554291, "reward_meter_std": 0.3436703681945801, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9904083013534546, "reward_repeat_soft_std": 0.012789538130164146, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.663899838924408, "reward_total_composite_std": 0.15209518373012543} {"timestamp_utc": "2026-04-12T22:40:36Z", "mode": "train", "global_step": 190, "epoch": 0.019085886489201405, "loss": 0.1071, "grad_norm": 13.938616752624512, "learning_rate": 9.427272727272728e-06, "num_tokens": 365233.0, "completions/mean_length": 57.125, "completions/min_length": 49.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.125, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.7486714720726013, "rewards/meter/std": 0.38221657276153564, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9878576993942261, "rewards/repeat_soft/std": 0.017484119161963463, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7150629162788391, "rewards/total_composite/std": 0.17185485363006592, "reward": 0.7150629162788391, "reward_std": 0.17185483872890472, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1873229295015335, "sampling/sampling_logp_difference/max": 1.6582221984863281, "sampling/importance_sampling_ratio/min": 0.19047731161117554, "sampling/importance_sampling_ratio/mean": 1.0213521718978882, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.981810376048088, "clip_ratio/low_mean": 0.0448082871735096, "clip_ratio/low_min": 0.0448082871735096, "clip_ratio/high_mean": 0.16760712210088968, "clip_ratio/high_max": 0.16760712210088968, "clip_ratio/region_mean": 0.21241540927439928, "reward_total_mean": 0.7150629162788391, "reward_meter_mean": 0.7486714720726013, "reward_meter_std": 0.38221657276153564, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9878576993942261, "reward_repeat_soft_std": 0.017484119161963463, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7150629162788391, "reward_total_composite_std": 0.17185485363006592} {"timestamp_utc": "2026-04-12T22:40:44Z", "mode": "train", "global_step": 191, "epoch": 0.0191863385233551, "loss": 0.0827, "grad_norm": 6.0124125480651855, "learning_rate": 9.424242424242425e-06, "num_tokens": 368166.0, "completions/mean_length": 181.625, "completions/min_length": 158.0, "completions/max_length": 211.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 181.625, "completions/min_terminated_length": 158.0, "completions/max_terminated_length": 211.0, "rewards/meter/mean": 0.9360668659210205, "rewards/meter/std": 0.11463984847068787, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9874351620674133, "rewards/repeat_soft/std": 0.011888244189321995, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.7757235765457153, "rewards/total_composite/std": 0.05351750925183296, "reward": 0.7757235765457153, "reward_std": 0.05351749435067177, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1810183823108673, "sampling/sampling_logp_difference/max": 1.7759523391723633, "sampling/importance_sampling_ratio/min": 0.16932211816310883, "sampling/importance_sampling_ratio/mean": 1.0242165327072144, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2983481884002686, "clip_ratio/low_mean": 0.06891964096575975, "clip_ratio/low_min": 0.06891964096575975, "clip_ratio/high_mean": 0.0929587259888649, "clip_ratio/high_max": 0.0929587259888649, "clip_ratio/region_mean": 0.16187836695462465, "reward_total_mean": 0.7757235765457153, "reward_meter_mean": 0.9360668659210205, "reward_meter_std": 0.11463984847068787, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9874351620674133, "reward_repeat_soft_std": 0.011888244189321995, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.7757235765457153, "reward_total_composite_std": 0.05351750925183296} {"timestamp_utc": "2026-04-12T22:40:50Z", "mode": "train", "global_step": 192, "epoch": 0.01928679055750879, "loss": 0.0511, "grad_norm": 20.492347717285156, "learning_rate": 9.421212121212122e-06, "num_tokens": 369877.0, "completions/mean_length": 42.875, "completions/min_length": 37.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.875, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.6187220215797424, "rewards/meter/std": 0.4059104025363922, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9923666715621948, "rewards/repeat_soft/std": 0.007227038033306599, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.6559115648269653, "rewards/total_composite/std": 0.18223440647125244, "reward": 0.6559115648269653, "reward_std": 0.18223440647125244, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17871472239494324, "sampling/sampling_logp_difference/max": 1.7018980979919434, "sampling/importance_sampling_ratio/min": 0.18233710527420044, "sampling/importance_sampling_ratio/mean": 1.0145092010498047, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3617488741874695, "clip_ratio/low_mean": 0.06380685605108738, "clip_ratio/low_min": 0.06380685605108738, "clip_ratio/high_mean": 0.12363456562161446, "clip_ratio/high_max": 0.12363456562161446, "clip_ratio/region_mean": 0.18744142167270184, "reward_total_mean": 0.6559115648269653, "reward_meter_mean": 0.6187220215797424, "reward_meter_std": 0.4059104025363922, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9923666715621948, "reward_repeat_soft_std": 0.007227038033306599, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.6559115648269653, "reward_total_composite_std": 0.18223440647125244} {"timestamp_utc": "2026-04-12T22:40:57Z", "mode": "train", "global_step": 193, "epoch": 0.019387242591662482, "loss": 0.0257, "grad_norm": 8.760064125061035, "learning_rate": 9.418181818181818e-06, "num_tokens": 372067.0, "completions/mean_length": 95.75, "completions/min_length": 89.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.75, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.7216939926147461, "rewards/meter/std": 0.38140353560447693, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9865915775299072, "rewards/repeat_soft/std": 0.015576460398733616, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.6802964806556702, "rewards/total_composite/std": 0.1582382768392563, "reward": 0.6802964806556702, "reward_std": 0.1582382768392563, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1715310662984848, "sampling/sampling_logp_difference/max": 1.0202453136444092, "sampling/importance_sampling_ratio/min": 0.37454289197921753, "sampling/importance_sampling_ratio/mean": 1.0401523113250732, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0904775112867355, "clip_ratio/low_mean": 0.06285456381738186, "clip_ratio/low_min": 0.06285456381738186, "clip_ratio/high_mean": 0.10084935091435909, "clip_ratio/high_max": 0.10084935091435909, "clip_ratio/region_mean": 0.16370391473174095, "reward_total_mean": 0.6802964806556702, "reward_meter_mean": 0.7216939926147461, "reward_meter_std": 0.38140353560447693, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9865915775299072, "reward_repeat_soft_std": 0.015576460398733616, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.6802964806556702, "reward_total_composite_std": 0.1582382768392563} {"timestamp_utc": "2026-04-12T22:41:03Z", "mode": "train", "global_step": 194, "epoch": 0.019487694625816173, "loss": 0.0545, "grad_norm": 9.997492790222168, "learning_rate": 9.415151515151515e-06, "num_tokens": 373859.0, "completions/mean_length": 67.0, "completions/min_length": 49.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9916137456893921, "rewards/meter/std": 0.006394678261131048, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.995553731918335, "rewards/repeat_soft/std": 0.0062927003018558025, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.8026565313339233, "rewards/total_composite/std": 0.02471695840358734, "reward": 0.8026565313339233, "reward_std": 0.02471696026623249, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18926693499088287, "sampling/sampling_logp_difference/max": 1.1049785614013672, "sampling/importance_sampling_ratio/min": 0.33121800422668457, "sampling/importance_sampling_ratio/mean": 1.0358859300613403, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3879302740097046, "clip_ratio/low_mean": 0.06041587330400944, "clip_ratio/low_min": 0.06041587330400944, "clip_ratio/high_mean": 0.12720229662954807, "clip_ratio/high_max": 0.12720229662954807, "clip_ratio/region_mean": 0.1876181699335575, "reward_total_mean": 0.8026565313339233, "reward_meter_mean": 0.9916137456893921, "reward_meter_std": 0.006394678261131048, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.995553731918335, "reward_repeat_soft_std": 0.0062927003018558025, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.8026565313339233, "reward_total_composite_std": 0.02471695840358734} {"timestamp_utc": "2026-04-12T22:41:10Z", "mode": "train", "global_step": 195, "epoch": 0.019588146659969864, "loss": -0.0807, "grad_norm": 10.878082275390625, "learning_rate": 9.412121212121212e-06, "num_tokens": 375657.0, "completions/mean_length": 65.75, "completions/min_length": 38.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.75, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.5816943645477295, "rewards/meter/std": 0.37206828594207764, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9994540810585022, "rewards/repeat_soft/std": 0.0010108178248628974, "rewards/judge_quality/mean": 0.36000001430511475, "rewards/judge_quality/std": 0.09165150672197342, "rewards/total_composite/mean": 0.6103328466415405, "rewards/total_composite/std": 0.19548235833644867, "reward": 0.6103328466415405, "reward_std": 0.19548232853412628, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2016046941280365, "sampling/sampling_logp_difference/max": 1.774024486541748, "sampling/importance_sampling_ratio/min": 0.16964885592460632, "sampling/importance_sampling_ratio/mean": 1.0383213758468628, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4231244772672653, "clip_ratio/low_mean": 0.08922833111137152, "clip_ratio/low_min": 0.08922833111137152, "clip_ratio/high_mean": 0.09207563661038876, "clip_ratio/high_max": 0.09207563661038876, "clip_ratio/region_mean": 0.18130396772176027, "reward_total_mean": 0.6103328466415405, "reward_meter_mean": 0.5816943645477295, "reward_meter_std": 0.37206828594207764, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9994540810585022, "reward_repeat_soft_std": 0.0010108178248628974, "reward_judge_quality_mean": 0.36000001430511475, "reward_judge_quality_std": 0.09165150672197342, "reward_total_composite_mean": 0.6103328466415405, "reward_total_composite_std": 0.19548235833644867} {"timestamp_utc": "2026-04-12T22:41:16Z", "mode": "train", "global_step": 196, "epoch": 0.019688598694123555, "loss": 0.0618, "grad_norm": 17.475666046142578, "learning_rate": 9.40909090909091e-06, "num_tokens": 377070.0, "completions/mean_length": 27.625, "completions/min_length": 22.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.625, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.4827464818954468, "rewards/meter/std": 0.43062275648117065, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.24833375215530396, "rewards/total_composite/mean": 0.6228609085083008, "rewards/total_composite/std": 0.22536708414554596, "reward": 0.6228609085083008, "reward_std": 0.22536706924438477, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17893603444099426, "sampling/sampling_logp_difference/max": 1.1922149658203125, "sampling/importance_sampling_ratio/min": 0.32511380314826965, "sampling/importance_sampling_ratio/mean": 1.019759178161621, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8978001922369003, "clip_ratio/low_mean": 0.09404377825558186, "clip_ratio/low_min": 0.09404377825558186, "clip_ratio/high_mean": 0.11062751803547144, "clip_ratio/high_max": 0.11062751803547144, "clip_ratio/region_mean": 0.2046712962910533, "reward_total_mean": 0.6228609085083008, "reward_meter_mean": 0.4827464818954468, "reward_meter_std": 0.43062275648117065, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.24833375215530396, "reward_total_composite_mean": 0.6228609085083008, "reward_total_composite_std": 0.22536708414554596} {"timestamp_utc": "2026-04-12T22:41:22Z", "mode": "train", "global_step": 197, "epoch": 0.019789050728277247, "loss": 0.1273, "grad_norm": 22.129459381103516, "learning_rate": 9.406060606060607e-06, "num_tokens": 378529.0, "completions/mean_length": 35.375, "completions/min_length": 30.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.375, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.8654494285583496, "rewards/meter/std": 0.34853827953338623, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.8025000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.8764522671699524, "rewards/total_composite/std": 0.20609977841377258, "reward": 0.8764522671699524, "reward_std": 0.2060997486114502, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15852470695972443, "sampling/sampling_logp_difference/max": 0.9438915252685547, "sampling/importance_sampling_ratio/min": 0.3891106843948364, "sampling/importance_sampling_ratio/mean": 1.011351466178894, "sampling/importance_sampling_ratio/max": 1.956284761428833, "entropy": 1.4904959127306938, "clip_ratio/low_mean": 0.04118217155337334, "clip_ratio/low_min": 0.04118217155337334, "clip_ratio/high_mean": 0.12981951143592596, "clip_ratio/high_max": 0.12981951143592596, "clip_ratio/region_mean": 0.1710016829892993, "reward_total_mean": 0.8764522671699524, "reward_meter_mean": 0.8654494285583496, "reward_meter_std": 0.34853827953338623, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.8025000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.8764522671699524, "reward_total_composite_std": 0.20609977841377258} {"timestamp_utc": "2026-04-12T22:41:29Z", "mode": "train", "global_step": 198, "epoch": 0.019889502762430938, "loss": 0.1145, "grad_norm": 16.026708602905273, "learning_rate": 9.403030303030304e-06, "num_tokens": 380541.0, "completions/mean_length": 68.5, "completions/min_length": 61.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.5, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.4582224190235138, "rewards/meter/std": 0.3012732267379761, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9965095520019531, "rewards/repeat_soft/std": 0.002199555281549692, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.6193510293960571, "rewards/total_composite/std": 0.18768593668937683, "reward": 0.6193510293960571, "reward_std": 0.18768592178821564, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22508522868156433, "sampling/sampling_logp_difference/max": 1.8852663040161133, "sampling/importance_sampling_ratio/min": 0.1517886370420456, "sampling/importance_sampling_ratio/mean": 0.9763094186782837, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.054028607904911, "clip_ratio/low_mean": 0.13065637461841106, "clip_ratio/low_min": 0.13065637461841106, "clip_ratio/high_mean": 0.06116020958870649, "clip_ratio/high_max": 0.06116020958870649, "clip_ratio/region_mean": 0.19181658420711756, "reward_total_mean": 0.6193510293960571, "reward_meter_mean": 0.4582224190235138, "reward_meter_std": 0.3012732267379761, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9965095520019531, "reward_repeat_soft_std": 0.002199555281549692, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.6193510293960571, "reward_total_composite_std": 0.18768593668937683} {"timestamp_utc": "2026-04-12T22:41:35Z", "mode": "train", "global_step": 199, "epoch": 0.019989954796584632, "loss": 0.0819, "grad_norm": 17.990541458129883, "learning_rate": 9.4e-06, "num_tokens": 381947.0, "completions/mean_length": 30.75, "completions/min_length": 26.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.75, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9828565120697021, "rewards/meter/std": 0.01875847764313221, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9604166746139526, "rewards/repeat_soft/std": 0.00589255103841424, "rewards/judge_quality/mean": 0.36375001072883606, "rewards/judge_quality/std": 0.13265825808048248, "rewards/total_composite/mean": 0.7974521517753601, "rewards/total_composite/std": 0.037255607545375824, "reward": 0.7974521517753601, "reward_std": 0.037255603820085526, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18850040435791016, "sampling/sampling_logp_difference/max": 1.4986425638198853, "sampling/importance_sampling_ratio/min": 0.223433256149292, "sampling/importance_sampling_ratio/mean": 1.0070288181304932, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.056455612182617, "clip_ratio/low_mean": 0.04545454680919647, "clip_ratio/low_min": 0.04545454680919647, "clip_ratio/high_mean": 0.178089483641088, "clip_ratio/high_max": 0.178089483641088, "clip_ratio/region_mean": 0.22354403045028448, "reward_total_mean": 0.7974521517753601, "reward_meter_mean": 0.9828565120697021, "reward_meter_std": 0.01875847764313221, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9604166746139526, "reward_repeat_soft_std": 0.00589255103841424, "reward_judge_quality_mean": 0.36375001072883606, "reward_judge_quality_std": 0.13265825808048248, "reward_total_composite_mean": 0.7974521517753601, "reward_total_composite_std": 0.037255607545375824} {"timestamp_utc": "2026-04-12T22:41:42Z", "mode": "train", "global_step": 200, "epoch": 0.020090406830738324, "loss": 0.0572, "grad_norm": 7.267741680145264, "learning_rate": 9.396969696969697e-06, "num_tokens": 384387.0, "completions/mean_length": 131.0, "completions/min_length": 111.0, "completions/max_length": 146.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.0, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 146.0, "rewards/meter/mean": 0.930094301700592, "rewards/meter/std": 0.12754081189632416, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.997650146484375, "rewards/repeat_soft/std": 0.0023139191325753927, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7815574407577515, "rewards/total_composite/std": 0.05706481263041496, "reward": 0.7815574407577515, "reward_std": 0.057064805179834366, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20789173245429993, "sampling/sampling_logp_difference/max": 2.3408408164978027, "sampling/importance_sampling_ratio/min": 0.09624667465686798, "sampling/importance_sampling_ratio/mean": 1.0487194061279297, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.758536249399185, "clip_ratio/low_mean": 0.05325818154960871, "clip_ratio/low_min": 0.05325818154960871, "clip_ratio/high_mean": 0.11834869347512722, "clip_ratio/high_max": 0.11834869347512722, "clip_ratio/region_mean": 0.17160687502473593, "reward_total_mean": 0.7815574407577515, "reward_meter_mean": 0.930094301700592, "reward_meter_std": 0.12754081189632416, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.997650146484375, "reward_repeat_soft_std": 0.0023139191325753927, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7815574407577515, "reward_total_composite_std": 0.05706481263041496} {"timestamp_utc": "2026-04-12T22:42:33Z", "mode": "eval", "global_step": 200, "epoch": 0.020090406830738324, "eval_loss": NaN, "eval_runtime": 51.3677, "eval_samples_per_second": 1.557, "eval_steps_per_second": 0.195, "eval_num_tokens": 384387.0, "eval_completions/mean_length": 98.4375, "eval_completions/min_length": 38.0, "eval_completions/max_length": 184.9, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 98.4375, "eval_completions/min_terminated_length": 38.0, "eval_completions/max_terminated_length": 184.9, "eval_rewards/meter/mean": 0.6527223944664001, "eval_rewards/meter/std": 0.3822064697742462, "eval_rewards/count_adherence/mean": 0.9860416531562806, "eval_rewards/count_adherence/std": 0.030216221511363984, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.07071067690849304, "eval_rewards/repeat_soft/mean": 0.9887738764286041, "eval_rewards/repeat_soft/std": 0.015243505884427577, "eval_rewards/judge_quality/mean": 0.3777499973773956, "eval_rewards/judge_quality/std": 0.13215117231011392, "eval_rewards/total_composite/mean": 0.6343437016010285, "eval_rewards/total_composite/std": 0.2018197938799858, "eval_reward": 0.6343437016010285, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.1359984040260315, "eval_sampling/sampling_logp_difference/max": 1.2542716979980468, "eval_sampling/importance_sampling_ratio/min": 0.28756721019744874, "eval_sampling/importance_sampling_ratio/mean": 1.0338929533958434, "eval_sampling/importance_sampling_ratio/max": 1.5515777111053466, "eval_entropy": 2.141538119316101, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6343437016010285, "eval_reward_meter_mean": 0.6527223944664001, "eval_reward_meter_std": 0.3822064697742462, "eval_reward_count_adherence_mean": 0.9860416531562806, "eval_reward_count_adherence_std": 0.030216221511363984, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.07071067690849304, "eval_reward_repeat_soft_mean": 0.9887738764286041, "eval_reward_repeat_soft_std": 0.015243505884427577, "eval_reward_judge_quality_mean": 0.3777499973773956, "eval_reward_judge_quality_std": 0.13215117231011392, "eval_reward_total_composite_mean": 0.6343437016010285, "eval_reward_total_composite_std": 0.2018197938799858} {"timestamp_utc": "2026-04-12T22:42:43Z", "mode": "train", "global_step": 201, "epoch": 0.020190858864892015, "loss": -0.0725, "grad_norm": 9.45182991027832, "learning_rate": 9.393939393939396e-06, "num_tokens": 386624.0, "completions/mean_length": 90.625, "completions/min_length": 53.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.625, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.7137892246246338, "rewards/meter/std": 0.38705575466156006, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9941529035568237, "rewards/repeat_soft/std": 0.004700822290033102, "rewards/judge_quality/mean": 0.30124998092651367, "rewards/judge_quality/std": 0.10398317128419876, "rewards/total_composite/mean": 0.6609954833984375, "rewards/total_composite/std": 0.19375428557395935, "reward": 0.6609954833984375, "reward_std": 0.19375428557395935, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1688535064458847, "sampling/sampling_logp_difference/max": 1.0816240310668945, "sampling/importance_sampling_ratio/min": 0.339044451713562, "sampling/importance_sampling_ratio/mean": 1.0338186025619507, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1685265600681305, "clip_ratio/low_mean": 0.04505446366965771, "clip_ratio/low_min": 0.04505446366965771, "clip_ratio/high_mean": 0.12919956259429455, "clip_ratio/high_max": 0.12919956259429455, "clip_ratio/region_mean": 0.17425402626395226, "reward_total_mean": 0.6609954833984375, "reward_meter_mean": 0.7137892246246338, "reward_meter_std": 0.38705575466156006, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9941529035568237, "reward_repeat_soft_std": 0.004700822290033102, "reward_judge_quality_mean": 0.30124998092651367, "reward_judge_quality_std": 0.10398317128419876, "reward_total_composite_mean": 0.6609954833984375, "reward_total_composite_std": 0.19375428557395935} {"timestamp_utc": "2026-04-12T22:42:51Z", "mode": "train", "global_step": 202, "epoch": 0.020291310899045706, "loss": -0.0316, "grad_norm": 8.221761703491211, "learning_rate": 9.390909090909092e-06, "num_tokens": 388803.0, "completions/mean_length": 100.375, "completions/min_length": 83.0, "completions/max_length": 115.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.375, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 115.0, "rewards/meter/mean": 0.9839462637901306, "rewards/meter/std": 0.02068553864955902, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.990378201007843, "rewards/repeat_soft/std": 0.010706717148423195, "rewards/judge_quality/mean": 0.32999998331069946, "rewards/judge_quality/std": 0.11747339367866516, "rewards/total_composite/mean": 0.5925955772399902, "rewards/total_composite/std": 0.36694273352622986, "reward": 0.5925955772399902, "reward_std": 0.36694273352622986, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1724347621202469, "sampling/sampling_logp_difference/max": 1.2036924362182617, "sampling/importance_sampling_ratio/min": 0.30008411407470703, "sampling/importance_sampling_ratio/mean": 1.0368696451187134, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.376349061727524, "clip_ratio/low_mean": 0.036620973609387875, "clip_ratio/low_min": 0.036620973609387875, "clip_ratio/high_mean": 0.10803817957639694, "clip_ratio/high_max": 0.10803817957639694, "clip_ratio/region_mean": 0.14465915318578482, "reward_total_mean": 0.5925955772399902, "reward_meter_mean": 0.9839462637901306, "reward_meter_std": 0.02068553864955902, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.990378201007843, "reward_repeat_soft_std": 0.010706717148423195, "reward_judge_quality_mean": 0.32999998331069946, "reward_judge_quality_std": 0.11747339367866516, "reward_total_composite_mean": 0.5925955772399902, "reward_total_composite_std": 0.36694273352622986} {"timestamp_utc": "2026-04-12T22:42:59Z", "mode": "train", "global_step": 203, "epoch": 0.020391762933199397, "loss": -0.0311, "grad_norm": 5.3910956382751465, "learning_rate": 9.387878787878789e-06, "num_tokens": 392045.0, "completions/mean_length": 198.25, "completions/min_length": 166.0, "completions/max_length": 229.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 198.25, "completions/min_terminated_length": 166.0, "completions/max_terminated_length": 229.0, "rewards/meter/mean": 0.8834912180900574, "rewards/meter/std": 0.21112480759620667, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.992335855960846, "rewards/repeat_soft/std": 0.006437603384256363, "rewards/judge_quality/mean": 0.2800000011920929, "rewards/judge_quality/std": 0.09304376691579819, "rewards/total_composite/mean": 0.7120546102523804, "rewards/total_composite/std": 0.07375628501176834, "reward": 0.7120546102523804, "reward_std": 0.07375627756118774, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18845759332180023, "sampling/sampling_logp_difference/max": 1.6991138458251953, "sampling/importance_sampling_ratio/min": 0.18284548819065094, "sampling/importance_sampling_ratio/mean": 1.0318385362625122, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.367601066827774, "clip_ratio/low_mean": 0.05321737378835678, "clip_ratio/low_min": 0.05321737378835678, "clip_ratio/high_mean": 0.11454436741769314, "clip_ratio/high_max": 0.11454436741769314, "clip_ratio/region_mean": 0.16776174120604992, "reward_total_mean": 0.7120546102523804, "reward_meter_mean": 0.8834912180900574, "reward_meter_std": 0.21112480759620667, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.992335855960846, "reward_repeat_soft_std": 0.006437603384256363, "reward_judge_quality_mean": 0.2800000011920929, "reward_judge_quality_std": 0.09304376691579819, "reward_total_composite_mean": 0.7120546102523804, "reward_total_composite_std": 0.07375628501176834} {"timestamp_utc": "2026-04-12T22:43:06Z", "mode": "train", "global_step": 204, "epoch": 0.020492214967353088, "loss": 0.0376, "grad_norm": 12.222681045532227, "learning_rate": 9.384848484848486e-06, "num_tokens": 393884.0, "completions/mean_length": 58.875, "completions/min_length": 44.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.875, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.8527190089225769, "rewards/meter/std": 0.23669300973415375, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9985201358795166, "rewards/repeat_soft/std": 0.002760024508461356, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.08166787773370743, "rewards/total_composite/mean": 0.5483859181404114, "rewards/total_composite/std": 0.35711216926574707, "reward": 0.5483859181404114, "reward_std": 0.3571121394634247, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19941037893295288, "sampling/sampling_logp_difference/max": 1.3351926803588867, "sampling/importance_sampling_ratio/min": 0.2631074786186218, "sampling/importance_sampling_ratio/mean": 1.0467816591262817, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5399347245693207, "clip_ratio/low_mean": 0.061958421021699905, "clip_ratio/low_min": 0.061958421021699905, "clip_ratio/high_mean": 0.11895784921944141, "clip_ratio/high_max": 0.11895784921944141, "clip_ratio/region_mean": 0.18091627024114132, "reward_total_mean": 0.5483859181404114, "reward_meter_mean": 0.8527190089225769, "reward_meter_std": 0.23669300973415375, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9985201358795166, "reward_repeat_soft_std": 0.002760024508461356, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.08166787773370743, "reward_total_composite_mean": 0.5483859181404114, "reward_total_composite_std": 0.35711216926574707} {"timestamp_utc": "2026-04-12T22:43:14Z", "mode": "train", "global_step": 205, "epoch": 0.02059266700150678, "loss": 0.0872, "grad_norm": 6.21863317489624, "learning_rate": 9.381818181818183e-06, "num_tokens": 396523.0, "completions/mean_length": 156.875, "completions/min_length": 139.0, "completions/max_length": 185.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 156.875, "completions/min_terminated_length": 139.0, "completions/max_terminated_length": 185.0, "rewards/meter/mean": 0.9915645718574524, "rewards/meter/std": 0.011944590136408806, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9956042766571045, "rewards/repeat_soft/std": 0.006392640061676502, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7949520349502563, "rewards/total_composite/std": 0.03068721294403076, "reward": 0.7949520349502563, "reward_std": 0.03068721294403076, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1586761325597763, "sampling/sampling_logp_difference/max": 1.254228115081787, "sampling/importance_sampling_ratio/min": 0.2852959632873535, "sampling/importance_sampling_ratio/mean": 1.029607892036438, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.133793383836746, "clip_ratio/low_mean": 0.06302296090871096, "clip_ratio/low_min": 0.06302296090871096, "clip_ratio/high_mean": 0.0820816420018673, "clip_ratio/high_max": 0.0820816420018673, "clip_ratio/region_mean": 0.14510460291057825, "reward_total_mean": 0.7949520349502563, "reward_meter_mean": 0.9915645718574524, "reward_meter_std": 0.011944590136408806, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9956042766571045, "reward_repeat_soft_std": 0.006392640061676502, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7949520349502563, "reward_total_composite_std": 0.03068721294403076} {"timestamp_utc": "2026-04-12T22:43:21Z", "mode": "train", "global_step": 206, "epoch": 0.02069311903566047, "loss": 0.0365, "grad_norm": 11.541095733642578, "learning_rate": 9.378787878787879e-06, "num_tokens": 398284.0, "completions/mean_length": 60.125, "completions/min_length": 43.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.125, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.4893374443054199, "rewards/meter/std": 0.3143778443336487, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9888759851455688, "rewards/repeat_soft/std": 0.02070561796426773, "rewards/judge_quality/mean": 0.4637500047683716, "rewards/judge_quality/std": 0.2930596172809601, "rewards/total_composite/mean": 0.6082144379615784, "rewards/total_composite/std": 0.19916348159313202, "reward": 0.6082144379615784, "reward_std": 0.19916348159313202, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20134250819683075, "sampling/sampling_logp_difference/max": 1.501279354095459, "sampling/importance_sampling_ratio/min": 0.22284488379955292, "sampling/importance_sampling_ratio/mean": 1.034473180770874, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1549899876117706, "clip_ratio/low_mean": 0.12721418589353561, "clip_ratio/low_min": 0.12721418589353561, "clip_ratio/high_mean": 0.08245599828660488, "clip_ratio/high_max": 0.08245599828660488, "clip_ratio/region_mean": 0.2096701841801405, "reward_total_mean": 0.6082144379615784, "reward_meter_mean": 0.4893374443054199, "reward_meter_std": 0.3143778443336487, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9888759851455688, "reward_repeat_soft_std": 0.02070561796426773, "reward_judge_quality_mean": 0.4637500047683716, "reward_judge_quality_std": 0.2930596172809601, "reward_total_composite_mean": 0.6082144379615784, "reward_total_composite_std": 0.19916348159313202} {"timestamp_utc": "2026-04-12T22:43:27Z", "mode": "train", "global_step": 207, "epoch": 0.020793571069814165, "loss": 0.0986, "grad_norm": 15.038732528686523, "learning_rate": 9.375757575757576e-06, "num_tokens": 400243.0, "completions/mean_length": 75.875, "completions/min_length": 57.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.875, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.5208646059036255, "rewards/meter/std": 0.3291780948638916, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9963343739509583, "rewards/repeat_soft/std": 0.0034947823733091354, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.16903086006641388, "rewards/total_composite/mean": 0.6850224733352661, "rewards/total_composite/std": 0.17327746748924255, "reward": 0.6850224733352661, "reward_std": 0.17327745258808136, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19959571957588196, "sampling/sampling_logp_difference/max": 1.8032383918762207, "sampling/importance_sampling_ratio/min": 0.16476444900035858, "sampling/importance_sampling_ratio/mean": 1.0130186080932617, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.139797255396843, "clip_ratio/low_mean": 0.1145754624158144, "clip_ratio/low_min": 0.1145754624158144, "clip_ratio/high_mean": 0.09555238112807274, "clip_ratio/high_max": 0.09555238112807274, "clip_ratio/region_mean": 0.21012784354388714, "reward_total_mean": 0.6850224733352661, "reward_meter_mean": 0.5208646059036255, "reward_meter_std": 0.3291780948638916, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9963343739509583, "reward_repeat_soft_std": 0.0034947823733091354, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.16903086006641388, "reward_total_composite_mean": 0.6850224733352661, "reward_total_composite_std": 0.17327746748924255} {"timestamp_utc": "2026-04-12T22:43:34Z", "mode": "train", "global_step": 208, "epoch": 0.020894023103967856, "loss": 0.0378, "grad_norm": 16.253944396972656, "learning_rate": 9.372727272727273e-06, "num_tokens": 401689.0, "completions/mean_length": 31.75, "completions/min_length": 28.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.75, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.6124137043952942, "rewards/meter/std": 0.40621063113212585, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.938178300857544, "rewards/repeat_soft/std": 0.027846481651067734, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.2499571591615677, "rewards/total_composite/mean": 0.6776540279388428, "rewards/total_composite/std": 0.19019806385040283, "reward": 0.6776540279388428, "reward_std": 0.19019807875156403, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1393977701663971, "sampling/sampling_logp_difference/max": 1.2818069458007812, "sampling/importance_sampling_ratio/min": 0.2775353491306305, "sampling/importance_sampling_ratio/mean": 1.0044059753417969, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0049168691039085, "clip_ratio/low_mean": 0.051893940195441246, "clip_ratio/low_min": 0.051893940195441246, "clip_ratio/high_mean": 0.07372742425650358, "clip_ratio/high_max": 0.07372742425650358, "clip_ratio/region_mean": 0.12562136445194483, "reward_total_mean": 0.6776540279388428, "reward_meter_mean": 0.6124137043952942, "reward_meter_std": 0.40621063113212585, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.938178300857544, "reward_repeat_soft_std": 0.027846481651067734, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.2499571591615677, "reward_total_composite_mean": 0.6776540279388428, "reward_total_composite_std": 0.19019806385040283} {"timestamp_utc": "2026-04-12T22:43:41Z", "mode": "train", "global_step": 209, "epoch": 0.020994475138121547, "loss": 0.0012, "grad_norm": 10.391623497009277, "learning_rate": 9.36969696969697e-06, "num_tokens": 403420.0, "completions/mean_length": 66.375, "completions/min_length": 44.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.375, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.6928176879882812, "rewards/meter/std": 0.4111105501651764, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9969114065170288, "rewards/repeat_soft/std": 0.004841444082558155, "rewards/judge_quality/mean": 0.4024999737739563, "rewards/judge_quality/std": 0.16446885466575623, "rewards/total_composite/mean": 0.6728341579437256, "rewards/total_composite/std": 0.23347313702106476, "reward": 0.6728341579437256, "reward_std": 0.23347312211990356, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17648614943027496, "sampling/sampling_logp_difference/max": 1.2686376571655273, "sampling/importance_sampling_ratio/min": 0.28121447563171387, "sampling/importance_sampling_ratio/mean": 1.027706503868103, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.167846515774727, "clip_ratio/low_mean": 0.02234848588705063, "clip_ratio/low_min": 0.02234848588705063, "clip_ratio/high_mean": 0.13506717514246702, "clip_ratio/high_max": 0.13506717514246702, "clip_ratio/region_mean": 0.15741566102951765, "reward_total_mean": 0.6728341579437256, "reward_meter_mean": 0.6928176879882812, "reward_meter_std": 0.4111105501651764, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9969114065170288, "reward_repeat_soft_std": 0.004841444082558155, "reward_judge_quality_mean": 0.4024999737739563, "reward_judge_quality_std": 0.16446885466575623, "reward_total_composite_mean": 0.6728341579437256, "reward_total_composite_std": 0.23347313702106476} {"timestamp_utc": "2026-04-12T22:43:48Z", "mode": "train", "global_step": 210, "epoch": 0.02109492717227524, "loss": 0.0511, "grad_norm": 7.208254337310791, "learning_rate": 9.366666666666668e-06, "num_tokens": 405820.0, "completions/mean_length": 135.0, "completions/min_length": 107.0, "completions/max_length": 144.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 135.0, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 144.0, "rewards/meter/mean": 0.9042514562606812, "rewards/meter/std": 0.19750821590423584, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9914119243621826, "rewards/repeat_soft/std": 0.0048458874225616455, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.756554365158081, "rewards/total_composite/std": 0.08106998354196548, "reward": 0.756554365158081, "reward_std": 0.08106997609138489, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16824375092983246, "sampling/sampling_logp_difference/max": 1.6585102081298828, "sampling/importance_sampling_ratio/min": 0.19042246043682098, "sampling/importance_sampling_ratio/mean": 1.0354307889938354, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.138040006160736, "clip_ratio/low_mean": 0.039393819868564606, "clip_ratio/low_min": 0.039393819868564606, "clip_ratio/high_mean": 0.13787824288010597, "clip_ratio/high_max": 0.13787824288010597, "clip_ratio/region_mean": 0.17727206274867058, "reward_total_mean": 0.756554365158081, "reward_meter_mean": 0.9042514562606812, "reward_meter_std": 0.19750821590423584, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9914119243621826, "reward_repeat_soft_std": 0.0048458874225616455, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.756554365158081, "reward_total_composite_std": 0.08106998354196548} {"timestamp_utc": "2026-04-12T22:43:54Z", "mode": "train", "global_step": 211, "epoch": 0.02119537920642893, "loss": 0.0704, "grad_norm": 14.57507038116455, "learning_rate": 9.363636363636365e-06, "num_tokens": 407333.0, "completions/mean_length": 29.125, "completions/min_length": 17.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.125, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.996258556842804, "rewards/meter/std": 0.002240557223558426, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.7976913452148438, "rewards/total_composite/std": 0.03349299728870392, "reward": 0.7976913452148438, "reward_std": 0.03349298983812332, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1744379997253418, "sampling/sampling_logp_difference/max": 1.1271677017211914, "sampling/importance_sampling_ratio/min": 0.3239494860172272, "sampling/importance_sampling_ratio/mean": 1.0234277248382568, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7621784657239914, "clip_ratio/low_mean": 0.10107611119747162, "clip_ratio/low_min": 0.10107611119747162, "clip_ratio/high_mean": 0.11261280253529549, "clip_ratio/high_max": 0.11261280253529549, "clip_ratio/region_mean": 0.2136889137327671, "reward_total_mean": 0.7976913452148438, "reward_meter_mean": 0.996258556842804, "reward_meter_std": 0.002240557223558426, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.7976913452148438, "reward_total_composite_std": 0.03349299728870392} {"timestamp_utc": "2026-04-12T22:44:01Z", "mode": "train", "global_step": 212, "epoch": 0.02129583124058262, "loss": 0.0435, "grad_norm": 11.12042236328125, "learning_rate": 9.36060606060606e-06, "num_tokens": 409120.0, "completions/mean_length": 65.375, "completions/min_length": 60.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.375, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9937935471534729, "rewards/meter/std": 0.008782095275819302, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9617397785186768, "rewards/repeat_soft/std": 0.044201161712408066, "rewards/judge_quality/mean": 0.4024999737739563, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.8141311407089233, "rewards/total_composite/std": 0.017935071140527725, "reward": 0.8141311407089233, "reward_std": 0.01793508231639862, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1533195525407791, "sampling/sampling_logp_difference/max": 1.0584931373596191, "sampling/importance_sampling_ratio/min": 0.3782862722873688, "sampling/importance_sampling_ratio/mean": 1.0483819246292114, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9591315537691116, "clip_ratio/low_mean": 0.027546045370399952, "clip_ratio/low_min": 0.027546045370399952, "clip_ratio/high_mean": 0.10376305785030127, "clip_ratio/high_max": 0.10376305785030127, "clip_ratio/region_mean": 0.13130910322070122, "reward_total_mean": 0.8141311407089233, "reward_meter_mean": 0.9937935471534729, "reward_meter_std": 0.008782095275819302, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9617397785186768, "reward_repeat_soft_std": 0.044201161712408066, "reward_judge_quality_mean": 0.4024999737739563, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.8141311407089233, "reward_total_composite_std": 0.017935071140527725} {"timestamp_utc": "2026-04-12T22:44:09Z", "mode": "train", "global_step": 213, "epoch": 0.021396283274736312, "loss": 0.0364, "grad_norm": 7.315183162689209, "learning_rate": 9.357575757575758e-06, "num_tokens": 411913.0, "completions/mean_length": 165.125, "completions/min_length": 145.0, "completions/max_length": 189.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 165.125, "completions/min_terminated_length": 145.0, "completions/max_terminated_length": 189.0, "rewards/meter/mean": 0.4947296977043152, "rewards/meter/std": 0.29927948117256165, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.996452808380127, "rewards/repeat_soft/std": 0.0023691589012742043, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.5641486644744873, "rewards/total_composite/std": 0.12801861763000488, "reward": 0.5641486644744873, "reward_std": 0.12801861763000488, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1791200041770935, "sampling/sampling_logp_difference/max": 1.5338325500488281, "sampling/importance_sampling_ratio/min": 0.21570737659931183, "sampling/importance_sampling_ratio/mean": 1.0269614458084106, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.175251916050911, "clip_ratio/low_mean": 0.1019281204789877, "clip_ratio/low_min": 0.1019281204789877, "clip_ratio/high_mean": 0.07958623021841049, "clip_ratio/high_max": 0.07958623021841049, "clip_ratio/region_mean": 0.18151435069739819, "reward_total_mean": 0.5641486644744873, "reward_meter_mean": 0.4947296977043152, "reward_meter_std": 0.29927948117256165, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.996452808380127, "reward_repeat_soft_std": 0.0023691589012742043, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.5641486644744873, "reward_total_composite_std": 0.12801861763000488} {"timestamp_utc": "2026-04-12T22:44:16Z", "mode": "train", "global_step": 214, "epoch": 0.021496735308890003, "loss": 0.053, "grad_norm": 15.698607444763184, "learning_rate": 9.354545454545455e-06, "num_tokens": 413620.0, "completions/mean_length": 54.375, "completions/min_length": 46.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.375, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.681374728679657, "rewards/meter/std": 0.29896000027656555, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9897869825363159, "rewards/repeat_soft/std": 0.01058239582926035, "rewards/judge_quality/mean": 0.41874998807907104, "rewards/judge_quality/std": 0.14574317634105682, "rewards/total_composite/mean": 0.6812223196029663, "rewards/total_composite/std": 0.12718971073627472, "reward": 0.6812223196029663, "reward_std": 0.1271897256374359, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17061147093772888, "sampling/sampling_logp_difference/max": 2.0034542083740234, "sampling/importance_sampling_ratio/min": 0.13486862182617188, "sampling/importance_sampling_ratio/mean": 1.0346450805664062, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.752282589673996, "clip_ratio/low_mean": 0.07384132593870163, "clip_ratio/low_min": 0.07384132593870163, "clip_ratio/high_mean": 0.09478327073156834, "clip_ratio/high_max": 0.09478327073156834, "clip_ratio/region_mean": 0.16862459667026997, "reward_total_mean": 0.6812223196029663, "reward_meter_mean": 0.681374728679657, "reward_meter_std": 0.29896000027656555, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9897869825363159, "reward_repeat_soft_std": 0.01058239582926035, "reward_judge_quality_mean": 0.41874998807907104, "reward_judge_quality_std": 0.14574317634105682, "reward_total_composite_mean": 0.6812223196029663, "reward_total_composite_std": 0.12718971073627472} {"timestamp_utc": "2026-04-12T22:44:22Z", "mode": "train", "global_step": 215, "epoch": 0.021597187343043698, "loss": 0.0168, "grad_norm": 9.017045021057129, "learning_rate": 9.351515151515152e-06, "num_tokens": 415817.0, "completions/mean_length": 98.625, "completions/min_length": 94.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.625, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.7990944385528564, "rewards/meter/std": 0.31399622559547424, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9968795776367188, "rewards/repeat_soft/std": 0.0017273214180022478, "rewards/judge_quality/mean": 0.5399999618530273, "rewards/judge_quality/std": 0.22245386242866516, "rewards/total_composite/mean": 0.7712804675102234, "rewards/total_composite/std": 0.17402377724647522, "reward": 0.7712804675102234, "reward_std": 0.17402377724647522, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1794540286064148, "sampling/sampling_logp_difference/max": 2.4753665924072266, "sampling/importance_sampling_ratio/min": 0.08413214236497879, "sampling/importance_sampling_ratio/mean": 1.023207426071167, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8376175910234451, "clip_ratio/low_mean": 0.03789222054183483, "clip_ratio/low_min": 0.03789222054183483, "clip_ratio/high_mean": 0.09544784668833017, "clip_ratio/high_max": 0.09544784668833017, "clip_ratio/region_mean": 0.133340067230165, "reward_total_mean": 0.7712804675102234, "reward_meter_mean": 0.7990944385528564, "reward_meter_std": 0.31399622559547424, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9968795776367188, "reward_repeat_soft_std": 0.0017273214180022478, "reward_judge_quality_mean": 0.5399999618530273, "reward_judge_quality_std": 0.22245386242866516, "reward_total_composite_mean": 0.7712804675102234, "reward_total_composite_std": 0.17402377724647522} {"timestamp_utc": "2026-04-12T22:44:29Z", "mode": "train", "global_step": 216, "epoch": 0.02169763937719739, "loss": 0.111, "grad_norm": 9.139263153076172, "learning_rate": 9.34848484848485e-06, "num_tokens": 418271.0, "completions/mean_length": 92.75, "completions/min_length": 76.0, "completions/max_length": 115.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.75, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 115.0, "rewards/meter/mean": 0.32860684394836426, "rewards/meter/std": 0.3178398013114929, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9882336854934692, "rewards/repeat_soft/std": 0.006412803195416927, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.23898595571517944, "rewards/total_composite/mean": 0.51219642162323, "rewards/total_composite/std": 0.19768090546131134, "reward": 0.51219642162323, "reward_std": 0.19768089056015015, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20194599032402039, "sampling/sampling_logp_difference/max": 2.3947439193725586, "sampling/importance_sampling_ratio/min": 0.09119603037834167, "sampling/importance_sampling_ratio/mean": 1.0125417709350586, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4624213576316833, "clip_ratio/low_mean": 0.13423257134854794, "clip_ratio/low_min": 0.13423257134854794, "clip_ratio/high_mean": 0.04835526458919048, "clip_ratio/high_max": 0.04835526458919048, "clip_ratio/region_mean": 0.18258783593773842, "reward_total_mean": 0.51219642162323, "reward_meter_mean": 0.32860684394836426, "reward_meter_std": 0.3178398013114929, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9882336854934692, "reward_repeat_soft_std": 0.006412803195416927, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.23898595571517944, "reward_total_composite_mean": 0.51219642162323, "reward_total_composite_std": 0.19768090546131134} {"timestamp_utc": "2026-04-12T22:44:38Z", "mode": "train", "global_step": 217, "epoch": 0.02179809141135108, "loss": 0.0786, "grad_norm": 4.809051036834717, "learning_rate": 9.345454545454547e-06, "num_tokens": 421467.0, "completions/mean_length": 200.5, "completions/min_length": 170.0, "completions/max_length": 217.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 200.5, "completions/min_terminated_length": 170.0, "completions/max_terminated_length": 217.0, "rewards/meter/mean": 0.9879785776138306, "rewards/meter/std": 0.020987501367926598, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9745301008224487, "rewards/repeat_soft/std": 0.019059404730796814, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.7737933397293091, "rewards/total_composite/std": 0.0372769795358181, "reward": 0.7737933397293091, "reward_std": 0.037276968359947205, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16808933019638062, "sampling/sampling_logp_difference/max": 1.7900724411010742, "sampling/importance_sampling_ratio/min": 0.16694806516170502, "sampling/importance_sampling_ratio/mean": 1.037062406539917, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.176947444677353, "clip_ratio/low_mean": 0.06853543221950531, "clip_ratio/low_min": 0.06853543221950531, "clip_ratio/high_mean": 0.06350381672382355, "clip_ratio/high_max": 0.06350381672382355, "clip_ratio/region_mean": 0.13203924894332886, "reward_total_mean": 0.7737933397293091, "reward_meter_mean": 0.9879785776138306, "reward_meter_std": 0.020987501367926598, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9745301008224487, "reward_repeat_soft_std": 0.019059404730796814, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.7737933397293091, "reward_total_composite_std": 0.0372769795358181} {"timestamp_utc": "2026-04-12T22:44:44Z", "mode": "train", "global_step": 218, "epoch": 0.02189854344550477, "loss": 0.0107, "grad_norm": 14.94848918914795, "learning_rate": 9.342424242424243e-06, "num_tokens": 423258.0, "completions/mean_length": 55.875, "completions/min_length": 36.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.875, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.7060632705688477, "rewards/meter/std": 0.38283491134643555, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9938966035842896, "rewards/repeat_soft/std": 0.012834918685257435, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.6942431330680847, "rewards/total_composite/std": 0.1732662171125412, "reward": 0.6942431330680847, "reward_std": 0.1732662469148636, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2042960226535797, "sampling/sampling_logp_difference/max": 1.2756810188293457, "sampling/importance_sampling_ratio/min": 0.2792407274246216, "sampling/importance_sampling_ratio/mean": 1.0386205911636353, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.241147205233574, "clip_ratio/low_mean": 0.08530469797551632, "clip_ratio/low_min": 0.08530469797551632, "clip_ratio/high_mean": 0.12841526325792074, "clip_ratio/high_max": 0.12841526325792074, "clip_ratio/region_mean": 0.21371996123343706, "reward_total_mean": 0.6942431330680847, "reward_meter_mean": 0.7060632705688477, "reward_meter_std": 0.38283491134643555, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9938966035842896, "reward_repeat_soft_std": 0.012834918685257435, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.6942431330680847, "reward_total_composite_std": 0.1732662171125412} {"timestamp_utc": "2026-04-12T22:44:51Z", "mode": "train", "global_step": 219, "epoch": 0.021998995479658463, "loss": 0.0196, "grad_norm": 11.719900131225586, "learning_rate": 9.33939393939394e-06, "num_tokens": 424945.0, "completions/mean_length": 62.875, "completions/min_length": 56.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.875, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.6097301244735718, "rewards/meter/std": 0.39728784561157227, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9982755780220032, "rewards/repeat_soft/std": 0.00300014391541481, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.6625810861587524, "rewards/total_composite/std": 0.21397148072719574, "reward": 0.6625810861587524, "reward_std": 0.21397146582603455, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17527978122234344, "sampling/sampling_logp_difference/max": 1.1357223987579346, "sampling/importance_sampling_ratio/min": 0.3211899995803833, "sampling/importance_sampling_ratio/mean": 1.044103980064392, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2035064846277237, "clip_ratio/low_mean": 0.041045167949050665, "clip_ratio/low_min": 0.041045167949050665, "clip_ratio/high_mean": 0.10810902435332537, "clip_ratio/high_max": 0.10810902435332537, "clip_ratio/region_mean": 0.14915419230237603, "reward_total_mean": 0.6625810861587524, "reward_meter_mean": 0.6097301244735718, "reward_meter_std": 0.39728784561157227, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9982755780220032, "reward_repeat_soft_std": 0.00300014391541481, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.6625810861587524, "reward_total_composite_std": 0.21397148072719574} {"timestamp_utc": "2026-04-12T22:44:57Z", "mode": "train", "global_step": 220, "epoch": 0.022099447513812154, "loss": 0.0964, "grad_norm": 17.653593063354492, "learning_rate": 9.336363636363637e-06, "num_tokens": 426647.0, "completions/mean_length": 47.75, "completions/min_length": 39.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.75, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.30497637391090393, "rewards/meter/std": 0.34212055802345276, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.988140344619751, "rewards/repeat_soft/std": 0.0252861138433218, "rewards/judge_quality/mean": 0.6487500071525574, "rewards/judge_quality/std": 0.2456151396036148, "rewards/total_composite/mean": 0.580678403377533, "rewards/total_composite/std": 0.18502292037010193, "reward": 0.580678403377533, "reward_std": 0.18502289056777954, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19322755932807922, "sampling/sampling_logp_difference/max": 2.5074644088745117, "sampling/importance_sampling_ratio/min": 0.08147456496953964, "sampling/importance_sampling_ratio/mean": 0.996791660785675, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0714217349886894, "clip_ratio/low_mean": 0.13291933946311474, "clip_ratio/low_min": 0.13291933946311474, "clip_ratio/high_mean": 0.07273874338716269, "clip_ratio/high_max": 0.07273874338716269, "clip_ratio/region_mean": 0.20565808285027742, "reward_total_mean": 0.580678403377533, "reward_meter_mean": 0.30497637391090393, "reward_meter_std": 0.34212055802345276, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.988140344619751, "reward_repeat_soft_std": 0.0252861138433218, "reward_judge_quality_mean": 0.6487500071525574, "reward_judge_quality_std": 0.2456151396036148, "reward_total_composite_mean": 0.580678403377533, "reward_total_composite_std": 0.18502292037010193} {"timestamp_utc": "2026-04-12T22:45:03Z", "mode": "train", "global_step": 221, "epoch": 0.022199899547965845, "loss": 0.1588, "grad_norm": 18.759233474731445, "learning_rate": 9.333333333333334e-06, "num_tokens": 428076.0, "completions/mean_length": 28.625, "completions/min_length": 13.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.625, "completions/min_terminated_length": 13.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.6394675970077515, "rewards/meter/std": 0.4421550929546356, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9472260475158691, "rewards/repeat_soft/std": 0.028762657195329666, "rewards/judge_quality/mean": 0.3787499964237213, "rewards/judge_quality/std": 0.2436295449733734, "rewards/total_composite/mean": 0.6461080312728882, "rewards/total_composite/std": 0.22164414823055267, "reward": 0.6461080312728882, "reward_std": 0.22164413332939148, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16225160658359528, "sampling/sampling_logp_difference/max": 1.0092346668243408, "sampling/importance_sampling_ratio/min": 0.3644978404045105, "sampling/importance_sampling_ratio/mean": 1.0396852493286133, "sampling/importance_sampling_ratio/max": 1.9327377080917358, "entropy": 1.852670133113861, "clip_ratio/low_mean": 0.030718903988599777, "clip_ratio/low_min": 0.030718903988599777, "clip_ratio/high_mean": 0.08502821158617735, "clip_ratio/high_max": 0.08502821158617735, "clip_ratio/region_mean": 0.11574711557477713, "reward_total_mean": 0.6461080312728882, "reward_meter_mean": 0.6394675970077515, "reward_meter_std": 0.4421550929546356, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9472260475158691, "reward_repeat_soft_std": 0.028762657195329666, "reward_judge_quality_mean": 0.3787499964237213, "reward_judge_quality_std": 0.2436295449733734, "reward_total_composite_mean": 0.6461080312728882, "reward_total_composite_std": 0.22164414823055267} {"timestamp_utc": "2026-04-12T22:45:09Z", "mode": "train", "global_step": 222, "epoch": 0.02230035158211954, "loss": 0.0564, "grad_norm": 14.19981575012207, "learning_rate": 9.33030303030303e-06, "num_tokens": 429787.0, "completions/mean_length": 50.875, "completions/min_length": 38.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.875, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.7463206052780151, "rewards/meter/std": 0.35328948497772217, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9845018982887268, "rewards/repeat_soft/std": 0.02211388200521469, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.7061694860458374, "rewards/total_composite/std": 0.17570172250270844, "reward": 0.7061694860458374, "reward_std": 0.17570170760154724, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17962686717510223, "sampling/sampling_logp_difference/max": 1.4296016693115234, "sampling/importance_sampling_ratio/min": 0.23940427601337433, "sampling/importance_sampling_ratio/mean": 1.044725775718689, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9504156559705734, "clip_ratio/low_mean": 0.04155969247221947, "clip_ratio/low_min": 0.04155969247221947, "clip_ratio/high_mean": 0.12596898153424263, "clip_ratio/high_max": 0.12596898153424263, "clip_ratio/region_mean": 0.1675286740064621, "reward_total_mean": 0.7061694860458374, "reward_meter_mean": 0.7463206052780151, "reward_meter_std": 0.35328948497772217, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9845018982887268, "reward_repeat_soft_std": 0.02211388200521469, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.7061694860458374, "reward_total_composite_std": 0.17570172250270844} {"timestamp_utc": "2026-04-12T22:45:16Z", "mode": "train", "global_step": 223, "epoch": 0.02240080361627323, "loss": 0.0257, "grad_norm": 7.719682693481445, "learning_rate": 9.327272727272729e-06, "num_tokens": 431970.0, "completions/mean_length": 100.875, "completions/min_length": 89.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.875, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.9940782785415649, "rewards/meter/std": 0.002984443213790655, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9923332929611206, "rewards/repeat_soft/std": 0.007202796172350645, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.7004891633987427, "rewards/total_composite/std": 0.2841547727584839, "reward": 0.7004891633987427, "reward_std": 0.2841547429561615, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18082353472709656, "sampling/sampling_logp_difference/max": 1.2701425552368164, "sampling/importance_sampling_ratio/min": 0.28079158067703247, "sampling/importance_sampling_ratio/mean": 1.0449841022491455, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.545370638370514, "clip_ratio/low_mean": 0.015776699408888817, "clip_ratio/low_min": 0.015776699408888817, "clip_ratio/high_mean": 0.1305677006021142, "clip_ratio/high_max": 0.1305677006021142, "clip_ratio/region_mean": 0.14634440001100302, "reward_total_mean": 0.7004891633987427, "reward_meter_mean": 0.9940782785415649, "reward_meter_std": 0.002984443213790655, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9923332929611206, "reward_repeat_soft_std": 0.007202796172350645, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.7004891633987427, "reward_total_composite_std": 0.2841547727584839} {"timestamp_utc": "2026-04-12T22:45:22Z", "mode": "train", "global_step": 224, "epoch": 0.022501255650426922, "loss": -0.0319, "grad_norm": 9.542694091796875, "learning_rate": 9.324242424242424e-06, "num_tokens": 433859.0, "completions/mean_length": 83.125, "completions/min_length": 63.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.125, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.954963207244873, "rewards/meter/std": 0.05485004186630249, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9916623830795288, "rewards/repeat_soft/std": 0.004886186216026545, "rewards/judge_quality/mean": 0.4475000202655792, "rewards/judge_quality/std": 0.12848013639450073, "rewards/total_composite/mean": 0.8131496906280518, "rewards/total_composite/std": 0.047446589916944504, "reward": 0.8131496906280518, "reward_std": 0.0474465973675251, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1831963062286377, "sampling/sampling_logp_difference/max": 1.5137519836425781, "sampling/importance_sampling_ratio/min": 0.22008268535137177, "sampling/importance_sampling_ratio/mean": 1.0295239686965942, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.975585401058197, "clip_ratio/low_mean": 0.05584507342427969, "clip_ratio/low_min": 0.05584507342427969, "clip_ratio/high_mean": 0.11094464547932148, "clip_ratio/high_max": 0.11094464547932148, "clip_ratio/region_mean": 0.16678971890360117, "reward_total_mean": 0.8131496906280518, "reward_meter_mean": 0.954963207244873, "reward_meter_std": 0.05485004186630249, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9916623830795288, "reward_repeat_soft_std": 0.004886186216026545, "reward_judge_quality_mean": 0.4475000202655792, "reward_judge_quality_std": 0.12848013639450073, "reward_total_composite_mean": 0.8131496906280518, "reward_total_composite_std": 0.047446589916944504} {"timestamp_utc": "2026-04-12T22:45:30Z", "mode": "train", "global_step": 225, "epoch": 0.022601707684580613, "loss": 0.0604, "grad_norm": 9.695013046264648, "learning_rate": 9.321212121212122e-06, "num_tokens": 436413.0, "completions/mean_length": 119.25, "completions/min_length": 94.0, "completions/max_length": 146.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.25, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 146.0, "rewards/meter/mean": 0.5250018835067749, "rewards/meter/std": 0.3770351707935333, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9959819912910461, "rewards/repeat_soft/std": 0.0034585981629788876, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.4833066463470459, "rewards/total_composite/std": 0.2536589801311493, "reward": 0.4833066463470459, "reward_std": 0.2536589801311493, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19236743450164795, "sampling/sampling_logp_difference/max": 1.2519326210021973, "sampling/importance_sampling_ratio/min": 0.2859516441822052, "sampling/importance_sampling_ratio/mean": 1.0415294170379639, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.555527776479721, "clip_ratio/low_mean": 0.06408403068780899, "clip_ratio/low_min": 0.06408403068780899, "clip_ratio/high_mean": 0.12092001549899578, "clip_ratio/high_max": 0.12092001549899578, "clip_ratio/region_mean": 0.18500404618680477, "reward_total_mean": 0.4833066463470459, "reward_meter_mean": 0.5250018835067749, "reward_meter_std": 0.3770351707935333, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9959819912910461, "reward_repeat_soft_std": 0.0034585981629788876, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.4833066463470459, "reward_total_composite_std": 0.2536589801311493} {"timestamp_utc": "2026-04-12T22:45:36Z", "mode": "train", "global_step": 226, "epoch": 0.022702159718734304, "loss": -0.0401, "grad_norm": 13.432271003723145, "learning_rate": 9.318181818181819e-06, "num_tokens": 437832.0, "completions/mean_length": 27.375, "completions/min_length": 22.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.375, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.7391682863235474, "rewards/meter/std": 0.24732904136180878, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9533430337905884, "rewards/repeat_soft/std": 0.0222688727080822, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7095850110054016, "rewards/total_composite/std": 0.11227390915155411, "reward": 0.7095850110054016, "reward_std": 0.11227390915155411, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1079363226890564, "sampling/sampling_logp_difference/max": 0.9618256092071533, "sampling/importance_sampling_ratio/min": 0.38219451904296875, "sampling/importance_sampling_ratio/mean": 1.016780138015747, "sampling/importance_sampling_ratio/max": 1.5922904014587402, "entropy": 0.8746723867952824, "clip_ratio/low_mean": 0.08646365813910961, "clip_ratio/low_min": 0.08646365813910961, "clip_ratio/high_mean": 0.07067448738962412, "clip_ratio/high_max": 0.07067448738962412, "clip_ratio/region_mean": 0.15713814552873373, "reward_total_mean": 0.7095850110054016, "reward_meter_mean": 0.7391682863235474, "reward_meter_std": 0.24732904136180878, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9533430337905884, "reward_repeat_soft_std": 0.0222688727080822, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7095850110054016, "reward_total_composite_std": 0.11227390915155411} {"timestamp_utc": "2026-04-12T22:45:43Z", "mode": "train", "global_step": 227, "epoch": 0.022802611752887995, "loss": 0.0181, "grad_norm": 6.291835308074951, "learning_rate": 9.315151515151516e-06, "num_tokens": 440362.0, "completions/mean_length": 140.25, "completions/min_length": 132.0, "completions/max_length": 151.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 140.25, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 151.0, "rewards/meter/mean": 0.8960759043693542, "rewards/meter/std": 0.20459839701652527, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.996003270149231, "rewards/repeat_soft/std": 0.004744499456137419, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.7533344626426697, "rewards/total_composite/std": 0.08933799713850021, "reward": 0.7533344626426697, "reward_std": 0.08933799713850021, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1747959405183792, "sampling/sampling_logp_difference/max": 2.201197624206543, "sampling/importance_sampling_ratio/min": 0.1106705367565155, "sampling/importance_sampling_ratio/mean": 1.0493431091308594, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3742400407791138, "clip_ratio/low_mean": 0.037852587178349495, "clip_ratio/low_min": 0.037852587178349495, "clip_ratio/high_mean": 0.11660830955952406, "clip_ratio/high_max": 0.11660830955952406, "clip_ratio/region_mean": 0.15446089673787355, "reward_total_mean": 0.7533344626426697, "reward_meter_mean": 0.8960759043693542, "reward_meter_std": 0.20459839701652527, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.996003270149231, "reward_repeat_soft_std": 0.004744499456137419, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.7533344626426697, "reward_total_composite_std": 0.08933799713850021} {"timestamp_utc": "2026-04-12T22:45:50Z", "mode": "train", "global_step": 228, "epoch": 0.022903063787041687, "loss": 0.024, "grad_norm": 10.584402084350586, "learning_rate": 9.312121212121212e-06, "num_tokens": 442195.0, "completions/mean_length": 59.125, "completions/min_length": 54.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.125, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9284141063690186, "rewards/meter/std": 0.11800399422645569, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9977113008499146, "rewards/repeat_soft/std": 0.0055434200912714005, "rewards/judge_quality/mean": 0.48500001430511475, "rewards/judge_quality/std": 0.2285982221364975, "rewards/total_composite/mean": 0.813057541847229, "rewards/total_composite/std": 0.0972953587770462, "reward": 0.813057541847229, "reward_std": 0.0972953587770462, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17132289707660675, "sampling/sampling_logp_difference/max": 1.5689716339111328, "sampling/importance_sampling_ratio/min": 0.2082592397928238, "sampling/importance_sampling_ratio/mean": 1.0282717943191528, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9079337120056152, "clip_ratio/low_mean": 0.08830753713846207, "clip_ratio/low_min": 0.08830753713846207, "clip_ratio/high_mean": 0.08729508332908154, "clip_ratio/high_max": 0.08729508332908154, "clip_ratio/region_mean": 0.1756026204675436, "reward_total_mean": 0.813057541847229, "reward_meter_mean": 0.9284141063690186, "reward_meter_std": 0.11800399422645569, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9977113008499146, "reward_repeat_soft_std": 0.0055434200912714005, "reward_judge_quality_mean": 0.48500001430511475, "reward_judge_quality_std": 0.2285982221364975, "reward_total_composite_mean": 0.813057541847229, "reward_total_composite_std": 0.0972953587770462} {"timestamp_utc": "2026-04-12T22:45:56Z", "mode": "train", "global_step": 229, "epoch": 0.023003515821195378, "loss": -0.0087, "grad_norm": 11.501396179199219, "learning_rate": 9.30909090909091e-06, "num_tokens": 443906.0, "completions/mean_length": 57.875, "completions/min_length": 35.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.875, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.978620171546936, "rewards/meter/std": 0.019945673644542694, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9931089878082275, "rewards/repeat_soft/std": 0.011518539860844612, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.8021900057792664, "rewards/total_composite/std": 0.03168172389268875, "reward": 0.8021900057792664, "reward_std": 0.03168172389268875, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16884812712669373, "sampling/sampling_logp_difference/max": 1.5398664474487305, "sampling/importance_sampling_ratio/min": 0.214409738779068, "sampling/importance_sampling_ratio/mean": 1.0520275831222534, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3012482821941376, "clip_ratio/low_mean": 0.07348901219666004, "clip_ratio/low_min": 0.07348901219666004, "clip_ratio/high_mean": 0.12493743561208248, "clip_ratio/high_max": 0.12493743561208248, "clip_ratio/region_mean": 0.19842644780874252, "reward_total_mean": 0.8021900057792664, "reward_meter_mean": 0.978620171546936, "reward_meter_std": 0.019945673644542694, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9931089878082275, "reward_repeat_soft_std": 0.011518539860844612, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.8021900057792664, "reward_total_composite_std": 0.03168172389268875} {"timestamp_utc": "2026-04-12T22:46:03Z", "mode": "train", "global_step": 230, "epoch": 0.023103967855349072, "loss": 0.1247, "grad_norm": 15.090785026550293, "learning_rate": 9.306060606060608e-06, "num_tokens": 445916.0, "completions/mean_length": 67.25, "completions/min_length": 62.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.25, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.6055946350097656, "rewards/meter/std": 0.43195900321006775, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.982811450958252, "rewards/repeat_soft/std": 0.016159115359187126, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.6404237747192383, "rewards/total_composite/std": 0.20140200853347778, "reward": 0.6404237747192383, "reward_std": 0.2014019936323166, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19653742015361786, "sampling/sampling_logp_difference/max": 2.1093668937683105, "sampling/importance_sampling_ratio/min": 0.12131474912166595, "sampling/importance_sampling_ratio/mean": 0.9851779937744141, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5822003185749054, "clip_ratio/low_mean": 0.0612277602776885, "clip_ratio/low_min": 0.0612277602776885, "clip_ratio/high_mean": 0.11418783850967884, "clip_ratio/high_max": 0.11418783850967884, "clip_ratio/region_mean": 0.17541559878736734, "reward_total_mean": 0.6404237747192383, "reward_meter_mean": 0.6055946350097656, "reward_meter_std": 0.43195900321006775, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.982811450958252, "reward_repeat_soft_std": 0.016159115359187126, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.6404237747192383, "reward_total_composite_std": 0.20140200853347778} {"timestamp_utc": "2026-04-12T22:46:11Z", "mode": "train", "global_step": 231, "epoch": 0.023204419889502764, "loss": 0.0272, "grad_norm": 9.040948867797852, "learning_rate": 9.303030303030303e-06, "num_tokens": 447745.0, "completions/mean_length": 64.625, "completions/min_length": 49.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.625, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.989717960357666, "rewards/meter/std": 0.01920364797115326, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9972941875457764, "rewards/repeat_soft/std": 0.004934768658131361, "rewards/judge_quality/mean": 0.5024999976158142, "rewards/judge_quality/std": 0.16705432534217834, "rewards/total_composite/mean": 0.8458524942398071, "rewards/total_composite/std": 0.052879635244607925, "reward": 0.8458524942398071, "reward_std": 0.05287962406873703, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15886950492858887, "sampling/sampling_logp_difference/max": 1.3513360023498535, "sampling/importance_sampling_ratio/min": 0.258894145488739, "sampling/importance_sampling_ratio/mean": 1.0313369035720825, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0065972805023193, "clip_ratio/low_mean": 0.08278274443000555, "clip_ratio/low_min": 0.08278274443000555, "clip_ratio/high_mean": 0.054113051854074, "clip_ratio/high_max": 0.054113051854074, "clip_ratio/region_mean": 0.13689579628407955, "reward_total_mean": 0.8458524942398071, "reward_meter_mean": 0.989717960357666, "reward_meter_std": 0.01920364797115326, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9972941875457764, "reward_repeat_soft_std": 0.004934768658131361, "reward_judge_quality_mean": 0.5024999976158142, "reward_judge_quality_std": 0.16705432534217834, "reward_total_composite_mean": 0.8458524942398071, "reward_total_composite_std": 0.052879635244607925} {"timestamp_utc": "2026-04-12T22:46:17Z", "mode": "train", "global_step": 232, "epoch": 0.023304871923656455, "loss": 0.0074, "grad_norm": 9.7235107421875, "learning_rate": 9.3e-06, "num_tokens": 449644.0, "completions/mean_length": 71.375, "completions/min_length": 67.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.375, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.8106899261474609, "rewards/meter/std": 0.32043975591659546, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9995647668838501, "rewards/repeat_soft/std": 0.0011147403856739402, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.8532669544219971, "rewards/total_composite/std": 0.1411026418209076, "reward": 0.8532669544219971, "reward_std": 0.1411026567220688, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16306047141551971, "sampling/sampling_logp_difference/max": 1.0975570678710938, "sampling/importance_sampling_ratio/min": 0.3336852490901947, "sampling/importance_sampling_ratio/mean": 1.0272496938705444, "sampling/importance_sampling_ratio/max": 1.9878556728363037, "entropy": 2.0106695145368576, "clip_ratio/low_mean": 0.05744114611297846, "clip_ratio/low_min": 0.05744114611297846, "clip_ratio/high_mean": 0.07862667366862297, "clip_ratio/high_max": 0.07862667366862297, "clip_ratio/region_mean": 0.13606781978160143, "reward_total_mean": 0.8532669544219971, "reward_meter_mean": 0.8106899261474609, "reward_meter_std": 0.32043975591659546, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9995647668838501, "reward_repeat_soft_std": 0.0011147403856739402, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.8532669544219971, "reward_total_composite_std": 0.1411026418209076} {"timestamp_utc": "2026-04-12T22:46:28Z", "mode": "train", "global_step": 233, "epoch": 0.023405323957810146, "loss": 0.0133, "grad_norm": 6.52683687210083, "learning_rate": 9.296969696969698e-06, "num_tokens": 452785.0, "completions/mean_length": 200.625, "completions/min_length": 155.0, "completions/max_length": 242.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 200.625, "completions/min_terminated_length": 155.0, "completions/max_terminated_length": 242.0, "rewards/meter/mean": 0.35411587357521057, "rewards/meter/std": 0.3424142599105835, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9926233291625977, "rewards/repeat_soft/std": 0.006160663906484842, "rewards/judge_quality/mean": 0.2887499928474426, "rewards/judge_quality/std": 0.11630470305681229, "rewards/total_composite/mean": 0.4702394902706146, "rewards/total_composite/std": 0.16173307597637177, "reward": 0.4702394902706146, "reward_std": 0.16173309087753296, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19548551738262177, "sampling/sampling_logp_difference/max": 1.9297070503234863, "sampling/importance_sampling_ratio/min": 0.14519073069095612, "sampling/importance_sampling_ratio/mean": 1.037513256072998, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.861027091741562, "clip_ratio/low_mean": 0.09871664829552174, "clip_ratio/low_min": 0.09871664829552174, "clip_ratio/high_mean": 0.08370550721883774, "clip_ratio/high_max": 0.08370550721883774, "clip_ratio/region_mean": 0.18242215551435947, "reward_total_mean": 0.4702394902706146, "reward_meter_mean": 0.35411587357521057, "reward_meter_std": 0.3424142599105835, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9926233291625977, "reward_repeat_soft_std": 0.006160663906484842, "reward_judge_quality_mean": 0.2887499928474426, "reward_judge_quality_std": 0.11630470305681229, "reward_total_composite_mean": 0.4702394902706146, "reward_total_composite_std": 0.16173307597637177} {"timestamp_utc": "2026-04-12T22:46:38Z", "mode": "train", "global_step": 234, "epoch": 0.023505775991963837, "loss": 0.0747, "grad_norm": 12.53782844543457, "learning_rate": 9.293939393939395e-06, "num_tokens": 454263.0, "completions/mean_length": 28.75, "completions/min_length": 23.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.75, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.6159642934799194, "rewards/meter/std": 0.4214301109313965, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.7649999856948853, "rewards/judge_quality/std": 0.2979933023452759, "rewards/total_composite/mean": 0.7529339790344238, "rewards/total_composite/std": 0.16097164154052734, "reward": 0.7529339790344238, "reward_std": 0.16097164154052734, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15589402616024017, "sampling/sampling_logp_difference/max": 1.3472023010253906, "sampling/importance_sampling_ratio/min": 0.25996655225753784, "sampling/importance_sampling_ratio/mean": 1.007192850112915, "sampling/importance_sampling_ratio/max": 1.6380813121795654, "entropy": 1.3930776715278625, "clip_ratio/low_mean": 0.04749920126050711, "clip_ratio/low_min": 0.04749920126050711, "clip_ratio/high_mean": 0.09122474864125252, "clip_ratio/high_max": 0.09122474864125252, "clip_ratio/region_mean": 0.13872394990175962, "reward_total_mean": 0.7529339790344238, "reward_meter_mean": 0.6159642934799194, "reward_meter_std": 0.4214301109313965, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.7649999856948853, "reward_judge_quality_std": 0.2979933023452759, "reward_total_composite_mean": 0.7529339790344238, "reward_total_composite_std": 0.16097164154052734} {"timestamp_utc": "2026-04-12T22:46:45Z", "mode": "train", "global_step": 235, "epoch": 0.023606228026117528, "loss": 0.1022, "grad_norm": 8.733047485351562, "learning_rate": 9.29090909090909e-06, "num_tokens": 456698.0, "completions/mean_length": 109.375, "completions/min_length": 85.0, "completions/max_length": 139.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.375, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.7656432390213013, "rewards/meter/std": 0.26865702867507935, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9946377873420715, "rewards/repeat_soft/std": 0.00715269148349762, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7136281728744507, "rewards/total_composite/std": 0.12620843946933746, "reward": 0.7136281728744507, "reward_std": 0.12620843946933746, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21398591995239258, "sampling/sampling_logp_difference/max": 1.3409252166748047, "sampling/importance_sampling_ratio/min": 0.2616035044193268, "sampling/importance_sampling_ratio/mean": 1.048514485359192, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.8638089895248413, "clip_ratio/low_mean": 0.07135637197643518, "clip_ratio/low_min": 0.07135637197643518, "clip_ratio/high_mean": 0.0928946677595377, "clip_ratio/high_max": 0.0928946677595377, "clip_ratio/region_mean": 0.16425103973597288, "reward_total_mean": 0.7136281728744507, "reward_meter_mean": 0.7656432390213013, "reward_meter_std": 0.26865702867507935, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9946377873420715, "reward_repeat_soft_std": 0.00715269148349762, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7136281728744507, "reward_total_composite_std": 0.12620843946933746} {"timestamp_utc": "2026-04-12T22:46:53Z", "mode": "train", "global_step": 236, "epoch": 0.02370668006027122, "loss": 0.034, "grad_norm": 11.12470531463623, "learning_rate": 9.28787878787879e-06, "num_tokens": 459255.0, "completions/mean_length": 125.625, "completions/min_length": 96.0, "completions/max_length": 144.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.625, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 144.0, "rewards/meter/mean": 0.7550405859947205, "rewards/meter/std": 0.19150368869304657, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9707974195480347, "rewards/repeat_soft/std": 0.057357292622327805, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.720973014831543, "rewards/total_composite/std": 0.09331081807613373, "reward": 0.720973014831543, "reward_std": 0.09331081062555313, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17728888988494873, "sampling/sampling_logp_difference/max": 2.9329891204833984, "sampling/importance_sampling_ratio/min": 0.0532376691699028, "sampling/importance_sampling_ratio/mean": 1.0145037174224854, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4931333884596825, "clip_ratio/low_mean": 0.07065259478986263, "clip_ratio/low_min": 0.07065259478986263, "clip_ratio/high_mean": 0.11190121248364449, "clip_ratio/high_max": 0.11190121248364449, "clip_ratio/region_mean": 0.18255380727350712, "reward_total_mean": 0.720973014831543, "reward_meter_mean": 0.7550405859947205, "reward_meter_std": 0.19150368869304657, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9707974195480347, "reward_repeat_soft_std": 0.057357292622327805, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.720973014831543, "reward_total_composite_std": 0.09331081807613373} {"timestamp_utc": "2026-04-12T22:46:59Z", "mode": "train", "global_step": 237, "epoch": 0.02380713209442491, "loss": 0.0483, "grad_norm": 7.966315269470215, "learning_rate": 9.284848484848485e-06, "num_tokens": 461151.0, "completions/mean_length": 83.0, "completions/min_length": 78.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.0, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.9735158085823059, "rewards/meter/std": 0.02861112169921398, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9988561868667603, "rewards/repeat_soft/std": 0.00124418328050524, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.20632413029670715, "rewards/total_composite/mean": 0.8120927810668945, "rewards/total_composite/std": 0.06512004882097244, "reward": 0.8120927810668945, "reward_std": 0.06512004137039185, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17589770257472992, "sampling/sampling_logp_difference/max": 2.1121904850006104, "sampling/importance_sampling_ratio/min": 0.12097268551588058, "sampling/importance_sampling_ratio/mean": 1.0163689851760864, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5606450587511063, "clip_ratio/low_mean": 0.07380836550146341, "clip_ratio/low_min": 0.07380836550146341, "clip_ratio/high_mean": 0.09933180175721645, "clip_ratio/high_max": 0.09933180175721645, "clip_ratio/region_mean": 0.17314016725867987, "reward_total_mean": 0.8120927810668945, "reward_meter_mean": 0.9735158085823059, "reward_meter_std": 0.02861112169921398, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9988561868667603, "reward_repeat_soft_std": 0.00124418328050524, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.20632413029670715, "reward_total_composite_mean": 0.8120927810668945, "reward_total_composite_std": 0.06512004882097244} {"timestamp_utc": "2026-04-12T22:47:08Z", "mode": "train", "global_step": 238, "epoch": 0.023907584128578605, "loss": 0.8318, "grad_norm": 13.187271118164062, "learning_rate": 9.281818181818183e-06, "num_tokens": 463139.0, "completions/mean_length": 91.5, "completions/min_length": 54.0, "completions/max_length": 321.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.5, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 321.0, "rewards/meter/mean": 0.7172454595565796, "rewards/meter/std": 0.4272644519805908, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9606211185455322, "rewards/repeat_soft/std": 0.09108290821313858, "rewards/judge_quality/mean": 0.42250001430511475, "rewards/judge_quality/std": 0.18108798563480377, "rewards/total_composite/mean": 0.6625491976737976, "rewards/total_composite/std": 0.30766403675079346, "reward": 0.6625491976737976, "reward_std": 0.30766406655311584, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21388381719589233, "sampling/sampling_logp_difference/max": 1.1890268325805664, "sampling/importance_sampling_ratio/min": 0.3045174479484558, "sampling/importance_sampling_ratio/mean": 1.0542694330215454, "sampling/importance_sampling_ratio/max": 1.8876317739486694, "entropy": 3.028022885322571, "clip_ratio/low_mean": 0.03792402148246765, "clip_ratio/low_min": 0.03792402148246765, "clip_ratio/high_mean": 0.1028674142435193, "clip_ratio/high_max": 0.1028674142435193, "clip_ratio/region_mean": 0.14079143572598696, "reward_total_mean": 0.6625491976737976, "reward_meter_mean": 0.7172454595565796, "reward_meter_std": 0.4272644519805908, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9606211185455322, "reward_repeat_soft_std": 0.09108290821313858, "reward_judge_quality_mean": 0.42250001430511475, "reward_judge_quality_std": 0.18108798563480377, "reward_total_composite_mean": 0.6625491976737976, "reward_total_composite_std": 0.30766403675079346} {"timestamp_utc": "2026-04-12T22:47:15Z", "mode": "train", "global_step": 239, "epoch": 0.024008036162732296, "loss": 0.0302, "grad_norm": 13.273502349853516, "learning_rate": 9.27878787878788e-06, "num_tokens": 464886.0, "completions/mean_length": 55.375, "completions/min_length": 42.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.375, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.6486506462097168, "rewards/meter/std": 0.39530149102211, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.996336817741394, "rewards/repeat_soft/std": 0.006857405882328749, "rewards/judge_quality/mean": 0.5024999976158142, "rewards/judge_quality/std": 0.2681550681591034, "rewards/total_composite/mean": 0.6922765374183655, "rewards/total_composite/std": 0.21277062594890594, "reward": 0.6922765374183655, "reward_std": 0.21277062594890594, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20548546314239502, "sampling/sampling_logp_difference/max": 1.4224605560302734, "sampling/importance_sampling_ratio/min": 0.24111999571323395, "sampling/importance_sampling_ratio/mean": 1.0364271402359009, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1999662667512894, "clip_ratio/low_mean": 0.05759575683623552, "clip_ratio/low_min": 0.05759575683623552, "clip_ratio/high_mean": 0.13599523156881332, "clip_ratio/high_max": 0.13599523156881332, "clip_ratio/region_mean": 0.19359098840504885, "reward_total_mean": 0.6922765374183655, "reward_meter_mean": 0.6486506462097168, "reward_meter_std": 0.39530149102211, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.996336817741394, "reward_repeat_soft_std": 0.006857405882328749, "reward_judge_quality_mean": 0.5024999976158142, "reward_judge_quality_std": 0.2681550681591034, "reward_total_composite_mean": 0.6922765374183655, "reward_total_composite_std": 0.21277062594890594} {"timestamp_utc": "2026-04-12T22:47:22Z", "mode": "train", "global_step": 240, "epoch": 0.024108488196885988, "loss": -0.0149, "grad_norm": 11.261079788208008, "learning_rate": 9.275757575757577e-06, "num_tokens": 466658.0, "completions/mean_length": 64.5, "completions/min_length": 56.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.5, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.8303419351577759, "rewards/meter/std": 0.2997992932796478, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9979175329208374, "rewards/repeat_soft/std": 0.004616478458046913, "rewards/judge_quality/mean": 0.35999998450279236, "rewards/judge_quality/std": 0.09165150672197342, "rewards/total_composite/mean": 0.7314456105232239, "rewards/total_composite/std": 0.15546607971191406, "reward": 0.7314456105232239, "reward_std": 0.15546607971191406, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.212741881608963, "sampling/sampling_logp_difference/max": 1.484156608581543, "sampling/importance_sampling_ratio/min": 0.22669345140457153, "sampling/importance_sampling_ratio/mean": 1.0417413711547852, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.943022519350052, "clip_ratio/low_mean": 0.038370998576283455, "clip_ratio/low_min": 0.038370998576283455, "clip_ratio/high_mean": 0.15426747128367424, "clip_ratio/high_max": 0.15426747128367424, "clip_ratio/region_mean": 0.1926384698599577, "reward_total_mean": 0.7314456105232239, "reward_meter_mean": 0.8303419351577759, "reward_meter_std": 0.2997992932796478, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9979175329208374, "reward_repeat_soft_std": 0.004616478458046913, "reward_judge_quality_mean": 0.35999998450279236, "reward_judge_quality_std": 0.09165150672197342, "reward_total_composite_mean": 0.7314456105232239, "reward_total_composite_std": 0.15546607971191406} {"timestamp_utc": "2026-04-12T22:47:31Z", "mode": "train", "global_step": 241, "epoch": 0.02420894023103968, "loss": -0.0398, "grad_norm": 10.336009979248047, "learning_rate": 9.272727272727273e-06, "num_tokens": 468638.0, "completions/mean_length": 84.5, "completions/min_length": 67.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.5, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.6151079535484314, "rewards/meter/std": 0.3954385221004486, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9956986904144287, "rewards/repeat_soft/std": 0.004514547996222973, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.25150617957115173, "rewards/total_composite/mean": 0.6834934949874878, "rewards/total_composite/std": 0.16889327764511108, "reward": 0.6834934949874878, "reward_std": 0.16889327764511108, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20427070558071136, "sampling/sampling_logp_difference/max": 1.922989845275879, "sampling/importance_sampling_ratio/min": 0.1461692899465561, "sampling/importance_sampling_ratio/mean": 1.037583589553833, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4143103063106537, "clip_ratio/low_mean": 0.04924237309023738, "clip_ratio/low_min": 0.04924237309023738, "clip_ratio/high_mean": 0.11960223410278559, "clip_ratio/high_max": 0.11960223410278559, "clip_ratio/region_mean": 0.16884460719302297, "reward_total_mean": 0.6834934949874878, "reward_meter_mean": 0.6151079535484314, "reward_meter_std": 0.3954385221004486, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9956986904144287, "reward_repeat_soft_std": 0.004514547996222973, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.25150617957115173, "reward_total_composite_mean": 0.6834934949874878, "reward_total_composite_std": 0.16889327764511108} {"timestamp_utc": "2026-04-12T22:47:37Z", "mode": "train", "global_step": 242, "epoch": 0.02430939226519337, "loss": 0.0356, "grad_norm": 28.52313995361328, "learning_rate": 9.26969696969697e-06, "num_tokens": 470079.0, "completions/mean_length": 23.125, "completions/min_length": 19.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.125, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.498305082321167, "rewards/meter/std": 0.46030333638191223, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.3512499928474426, "rewards/judge_quality/std": 0.11630470305681229, "rewards/total_composite/mean": 0.5758622884750366, "rewards/total_composite/std": 0.23587921261787415, "reward": 0.5758622884750366, "reward_std": 0.23587918281555176, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14344589412212372, "sampling/sampling_logp_difference/max": 2.1663403511047363, "sampling/importance_sampling_ratio/min": 0.11459623277187347, "sampling/importance_sampling_ratio/mean": 0.9992484450340271, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8322795741260052, "clip_ratio/low_mean": 0.05294940201565623, "clip_ratio/low_min": 0.05294940201565623, "clip_ratio/high_mean": 0.038419914431869984, "clip_ratio/high_max": 0.038419914431869984, "clip_ratio/region_mean": 0.09136931644752622, "reward_total_mean": 0.5758622884750366, "reward_meter_mean": 0.498305082321167, "reward_meter_std": 0.46030333638191223, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.3512499928474426, "reward_judge_quality_std": 0.11630470305681229, "reward_total_composite_mean": 0.5758622884750366, "reward_total_composite_std": 0.23587921261787415} {"timestamp_utc": "2026-04-12T22:47:44Z", "mode": "train", "global_step": 243, "epoch": 0.02440984429934706, "loss": 0.0722, "grad_norm": 10.151678085327148, "learning_rate": 9.266666666666667e-06, "num_tokens": 472070.0, "completions/mean_length": 85.875, "completions/min_length": 78.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.875, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.5651149749755859, "rewards/meter/std": 0.4197004735469818, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.991955578327179, "rewards/repeat_soft/std": 0.008437552489340305, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.6167473196983337, "rewards/total_composite/std": 0.19320012629032135, "reward": 0.6167473196983337, "reward_std": 0.19320012629032135, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18541502952575684, "sampling/sampling_logp_difference/max": 1.4706039428710938, "sampling/importance_sampling_ratio/min": 0.22978666424751282, "sampling/importance_sampling_ratio/mean": 1.038940191268921, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.329676851630211, "clip_ratio/low_mean": 0.08723794110119343, "clip_ratio/low_min": 0.08723794110119343, "clip_ratio/high_mean": 0.08647382818162441, "clip_ratio/high_max": 0.08647382818162441, "clip_ratio/region_mean": 0.17371176928281784, "reward_total_mean": 0.6167473196983337, "reward_meter_mean": 0.5651149749755859, "reward_meter_std": 0.4197004735469818, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.991955578327179, "reward_repeat_soft_std": 0.008437552489340305, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.6167473196983337, "reward_total_composite_std": 0.19320012629032135} {"timestamp_utc": "2026-04-12T22:47:51Z", "mode": "train", "global_step": 244, "epoch": 0.024510296333500752, "loss": 0.0696, "grad_norm": 11.594178199768066, "learning_rate": 9.263636363636364e-06, "num_tokens": 473719.0, "completions/mean_length": 58.125, "completions/min_length": 46.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.125, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.24202539026737213, "rewards/meter/std": 0.2553742825984955, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9992761611938477, "rewards/repeat_soft/std": 0.001410387922078371, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.0843462198972702, "rewards/total_composite/mean": 0.4743390679359436, "rewards/total_composite/std": 0.11051017045974731, "reward": 0.4743390679359436, "reward_std": 0.11051014810800552, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20545533299446106, "sampling/sampling_logp_difference/max": 1.0835037231445312, "sampling/importance_sampling_ratio/min": 0.3384077548980713, "sampling/importance_sampling_ratio/mean": 1.0498751401901245, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.7120381891727448, "clip_ratio/low_mean": 0.09280562400817871, "clip_ratio/low_min": 0.09280562400817871, "clip_ratio/high_mean": 0.06330752745270729, "clip_ratio/high_max": 0.06330752745270729, "clip_ratio/region_mean": 0.156113151460886, "reward_total_mean": 0.4743390679359436, "reward_meter_mean": 0.24202539026737213, "reward_meter_std": 0.2553742825984955, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9992761611938477, "reward_repeat_soft_std": 0.001410387922078371, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.0843462198972702, "reward_total_composite_mean": 0.4743390679359436, "reward_total_composite_std": 0.11051017045974731} {"timestamp_utc": "2026-04-12T22:47:58Z", "mode": "train", "global_step": 245, "epoch": 0.024610748367654443, "loss": 0.0455, "grad_norm": 12.824295043945312, "learning_rate": 9.260606060606062e-06, "num_tokens": 475562.0, "completions/mean_length": 56.375, "completions/min_length": 43.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.375, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.783051609992981, "rewards/meter/std": 0.28790801763534546, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9996891021728516, "rewards/repeat_soft/std": 0.0006069060764275491, "rewards/judge_quality/mean": 0.36374998092651367, "rewards/judge_quality/std": 0.09500939399003983, "rewards/total_composite/mean": 0.6594811677932739, "rewards/total_composite/std": 0.2779873311519623, "reward": 0.6594811677932739, "reward_std": 0.2779873311519623, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20442284643650055, "sampling/sampling_logp_difference/max": 1.208592414855957, "sampling/importance_sampling_ratio/min": 0.29861733317375183, "sampling/importance_sampling_ratio/mean": 1.0455328226089478, "sampling/importance_sampling_ratio/max": 1.9741625785827637, "entropy": 2.409978613257408, "clip_ratio/low_mean": 0.048394243232905865, "clip_ratio/low_min": 0.048394243232905865, "clip_ratio/high_mean": 0.136877179145813, "clip_ratio/high_max": 0.136877179145813, "clip_ratio/region_mean": 0.18527142237871885, "reward_total_mean": 0.6594811677932739, "reward_meter_mean": 0.783051609992981, "reward_meter_std": 0.28790801763534546, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9996891021728516, "reward_repeat_soft_std": 0.0006069060764275491, "reward_judge_quality_mean": 0.36374998092651367, "reward_judge_quality_std": 0.09500939399003983, "reward_total_composite_mean": 0.6594811677932739, "reward_total_composite_std": 0.2779873311519623} {"timestamp_utc": "2026-04-12T22:48:08Z", "mode": "train", "global_step": 246, "epoch": 0.024711200401808138, "loss": -0.0128, "grad_norm": 5.772524833679199, "learning_rate": 9.257575757575759e-06, "num_tokens": 478739.0, "completions/mean_length": 201.125, "completions/min_length": 147.0, "completions/max_length": 269.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 201.125, "completions/min_terminated_length": 147.0, "completions/max_terminated_length": 269.0, "rewards/meter/mean": 0.43260225653648376, "rewards/meter/std": 0.3558891713619232, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.10681164264678955, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9973302483558655, "rewards/repeat_soft/std": 0.003060556249693036, "rewards/judge_quality/mean": 0.25874999165534973, "rewards/judge_quality/std": 0.07395702600479126, "rewards/total_composite/mean": 0.4241829514503479, "rewards/total_composite/std": 0.23961442708969116, "reward": 0.4241829514503479, "reward_std": 0.23961441218852997, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21506229043006897, "sampling/sampling_logp_difference/max": 1.452444076538086, "sampling/importance_sampling_ratio/min": 0.23399768769741058, "sampling/importance_sampling_ratio/mean": 1.0483766794204712, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.610942929983139, "clip_ratio/low_mean": 0.05695440340787172, "clip_ratio/low_min": 0.05695440340787172, "clip_ratio/high_mean": 0.12621658854186535, "clip_ratio/high_max": 0.12621658854186535, "clip_ratio/region_mean": 0.18317099194973707, "reward_total_mean": 0.4241829514503479, "reward_meter_mean": 0.43260225653648376, "reward_meter_std": 0.3558891713619232, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.10681164264678955, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9973302483558655, "reward_repeat_soft_std": 0.003060556249693036, "reward_judge_quality_mean": 0.25874999165534973, "reward_judge_quality_std": 0.07395702600479126, "reward_total_composite_mean": 0.4241829514503479, "reward_total_composite_std": 0.23961442708969116} {"timestamp_utc": "2026-04-12T22:48:16Z", "mode": "train", "global_step": 247, "epoch": 0.02481165243596183, "loss": -0.0354, "grad_norm": 9.905174255371094, "learning_rate": 9.254545454545454e-06, "num_tokens": 480554.0, "completions/mean_length": 66.875, "completions/min_length": 50.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.875, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.3794197738170624, "rewards/meter/std": 0.2575729191303253, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9990973472595215, "rewards/repeat_soft/std": 0.0019004953792318702, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.5402736067771912, "rewards/total_composite/std": 0.12342807650566101, "reward": 0.5402736067771912, "reward_std": 0.12342805415391922, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20863491296768188, "sampling/sampling_logp_difference/max": 1.5102834701538086, "sampling/importance_sampling_ratio/min": 0.22084736824035645, "sampling/importance_sampling_ratio/mean": 1.0235635042190552, "sampling/importance_sampling_ratio/max": 1.8423658609390259, "entropy": 2.909391537308693, "clip_ratio/low_mean": 0.07251503877341747, "clip_ratio/low_min": 0.07251503877341747, "clip_ratio/high_mean": 0.12647932302206755, "clip_ratio/high_max": 0.12647932302206755, "clip_ratio/region_mean": 0.19899436179548502, "reward_total_mean": 0.5402736067771912, "reward_meter_mean": 0.3794197738170624, "reward_meter_std": 0.2575729191303253, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9990973472595215, "reward_repeat_soft_std": 0.0019004953792318702, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.5402736067771912, "reward_total_composite_std": 0.12342807650566101} {"timestamp_utc": "2026-04-12T22:48:23Z", "mode": "train", "global_step": 248, "epoch": 0.02491210447011552, "loss": -0.0216, "grad_norm": 8.936887741088867, "learning_rate": 9.251515151515152e-06, "num_tokens": 482600.0, "completions/mean_length": 95.75, "completions/min_length": 64.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.75, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.413760244846344, "rewards/meter/std": 0.2599477469921112, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9972332715988159, "rewards/repeat_soft/std": 0.003570000873878598, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.5555404424667358, "rewards/total_composite/std": 0.12218713760375977, "reward": 0.5555404424667358, "reward_std": 0.12218714505434036, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22171355783939362, "sampling/sampling_logp_difference/max": 1.506246566772461, "sampling/importance_sampling_ratio/min": 0.2217407077550888, "sampling/importance_sampling_ratio/mean": 1.0438802242279053, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.608716830611229, "clip_ratio/low_mean": 0.12864871509373188, "clip_ratio/low_min": 0.12864871509373188, "clip_ratio/high_mean": 0.06282605789601803, "clip_ratio/high_max": 0.06282605789601803, "clip_ratio/region_mean": 0.1914747729897499, "reward_total_mean": 0.5555404424667358, "reward_meter_mean": 0.413760244846344, "reward_meter_std": 0.2599477469921112, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9972332715988159, "reward_repeat_soft_std": 0.003570000873878598, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.5555404424667358, "reward_total_composite_std": 0.12218713760375977} {"timestamp_utc": "2026-04-12T22:48:29Z", "mode": "train", "global_step": 249, "epoch": 0.02501255650426921, "loss": 0.0716, "grad_norm": 18.126298904418945, "learning_rate": 9.248484848484849e-06, "num_tokens": 484116.0, "completions/mean_length": 44.5, "completions/min_length": 40.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.5, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.49406689405441284, "rewards/meter/std": 0.3191210925579071, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9943512678146362, "rewards/repeat_soft/std": 0.008807245641946793, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.6540151834487915, "rewards/total_composite/std": 0.17928293347358704, "reward": 0.6540151834487915, "reward_std": 0.17928291857242584, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1901768445968628, "sampling/sampling_logp_difference/max": 1.7334613800048828, "sampling/importance_sampling_ratio/min": 0.1766718178987503, "sampling/importance_sampling_ratio/mean": 1.0153061151504517, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0152252092957497, "clip_ratio/low_mean": 0.07058080844581127, "clip_ratio/low_min": 0.07058080844581127, "clip_ratio/high_mean": 0.09148098807781935, "clip_ratio/high_max": 0.09148098807781935, "clip_ratio/region_mean": 0.16206179652363062, "reward_total_mean": 0.6540151834487915, "reward_meter_mean": 0.49406689405441284, "reward_meter_std": 0.3191210925579071, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9943512678146362, "reward_repeat_soft_std": 0.008807245641946793, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.6540151834487915, "reward_total_composite_std": 0.17928293347358704} {"timestamp_utc": "2026-04-12T22:48:37Z", "mode": "train", "global_step": 250, "epoch": 0.025113008538422903, "loss": 0.0403, "grad_norm": 6.014837265014648, "learning_rate": 9.245454545454546e-06, "num_tokens": 487259.0, "completions/mean_length": 186.875, "completions/min_length": 132.0, "completions/max_length": 213.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 186.875, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 213.0, "rewards/meter/mean": 0.6916961073875427, "rewards/meter/std": 0.2751163840293884, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.994820773601532, "rewards/repeat_soft/std": 0.005338526330888271, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.6376203298568726, "rewards/total_composite/std": 0.12483447045087814, "reward": 0.6376203298568726, "reward_std": 0.12483445554971695, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2097940295934677, "sampling/sampling_logp_difference/max": 1.2559013366699219, "sampling/importance_sampling_ratio/min": 0.28481900691986084, "sampling/importance_sampling_ratio/mean": 1.0570787191390991, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.3793243765830994, "clip_ratio/low_mean": 0.07245279382914305, "clip_ratio/low_min": 0.07245279382914305, "clip_ratio/high_mean": 0.08382073976099491, "clip_ratio/high_max": 0.08382073976099491, "clip_ratio/region_mean": 0.15627353359013796, "reward_total_mean": 0.6376203298568726, "reward_meter_mean": 0.6916961073875427, "reward_meter_std": 0.2751163840293884, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.994820773601532, "reward_repeat_soft_std": 0.005338526330888271, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.6376203298568726, "reward_total_composite_std": 0.12483447045087814} {"timestamp_utc": "2026-04-12T22:49:38Z", "mode": "eval", "global_step": 250, "epoch": 0.025113008538422903, "eval_loss": NaN, "eval_runtime": 60.9051, "eval_samples_per_second": 1.314, "eval_steps_per_second": 0.164, "eval_num_tokens": 487259.0, "eval_completions/mean_length": 105.6375, "eval_completions/min_length": 41.1, "eval_completions/max_length": 262.4, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 95.37321548461914, "eval_completions/min_terminated_length": 41.1, "eval_completions/max_terminated_length": 194.1, "eval_rewards/meter/mean": 0.4890248000621796, "eval_rewards/meter/std": 0.3502065628767014, "eval_rewards/count_adherence/mean": 0.944166648387909, "eval_rewards/count_adherence/std": 0.12668615095317365, "eval_rewards/hard_gate/mean": 0.925, "eval_rewards/hard_gate/std": 0.21213203072547912, "eval_rewards/repeat_soft/mean": 0.993763518333435, "eval_rewards/repeat_soft/std": 0.009745006239973009, "eval_rewards/judge_quality/mean": 0.35987499058246614, "eval_rewards/judge_quality/std": 0.09623442320153117, "eval_rewards/total_composite/mean": 0.5474494695663452, "eval_rewards/total_composite/std": 0.2150147885084152, "eval_reward": 0.5474494695663452, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.1656752794981003, "eval_sampling/sampling_logp_difference/max": 1.264777946472168, "eval_sampling/importance_sampling_ratio/min": 0.28500361144542696, "eval_sampling/importance_sampling_ratio/mean": 1.043656599521637, "eval_sampling/importance_sampling_ratio/max": 1.588713538646698, "eval_entropy": 2.825564813613892, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5474494695663452, "eval_reward_meter_mean": 0.4890248000621796, "eval_reward_meter_std": 0.3502065628767014, "eval_reward_count_adherence_mean": 0.944166648387909, "eval_reward_count_adherence_std": 0.12668615095317365, "eval_reward_hard_gate_mean": 0.925, "eval_reward_hard_gate_std": 0.21213203072547912, "eval_reward_repeat_soft_mean": 0.993763518333435, "eval_reward_repeat_soft_std": 0.009745006239973009, "eval_reward_judge_quality_mean": 0.35987499058246614, "eval_reward_judge_quality_std": 0.09623442320153117, "eval_reward_total_composite_mean": 0.5474494695663452, "eval_reward_total_composite_std": 0.2150147885084152} {"timestamp_utc": "2026-04-12T22:49:52Z", "mode": "train", "global_step": 251, "epoch": 0.025213460572576594, "loss": 0.0007, "grad_norm": 13.189837455749512, "learning_rate": 9.242424242424244e-06, "num_tokens": 488928.0, "completions/mean_length": 52.625, "completions/min_length": 45.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.625, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.58943110704422, "rewards/meter/std": 0.37848860025405884, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 1.0, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.41874998807907104, "rewards/judge_quality/std": 0.23805385828018188, "rewards/total_composite/mean": 0.6025833487510681, "rewards/total_composite/std": 0.289580374956131, "reward": 0.6025833487510681, "reward_std": 0.2895803451538086, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21697130799293518, "sampling/sampling_logp_difference/max": 1.1746044158935547, "sampling/importance_sampling_ratio/min": 0.30894115567207336, "sampling/importance_sampling_ratio/mean": 1.0449841022491455, "sampling/importance_sampling_ratio/max": 1.9077218770980835, "entropy": 2.7707096934318542, "clip_ratio/low_mean": 0.08581154700368643, "clip_ratio/low_min": 0.08581154700368643, "clip_ratio/high_mean": 0.08794190362095833, "clip_ratio/high_max": 0.08794190362095833, "clip_ratio/region_mean": 0.17375345062464476, "reward_total_mean": 0.6025833487510681, "reward_meter_mean": 0.58943110704422, "reward_meter_std": 0.37848860025405884, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 1.0, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.41874998807907104, "reward_judge_quality_std": 0.23805385828018188, "reward_total_composite_mean": 0.6025833487510681, "reward_total_composite_std": 0.289580374956131} {"timestamp_utc": "2026-04-12T22:50:04Z", "mode": "train", "global_step": 252, "epoch": 0.025313912606730285, "loss": 0.0181, "grad_norm": 4.305587291717529, "learning_rate": 9.23939393939394e-06, "num_tokens": 490992.0, "completions/mean_length": 154.0, "completions/min_length": 67.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 102.85714721679688, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 223.0, "rewards/meter/mean": 0.19669239223003387, "rewards/meter/std": 0.23418445885181427, "rewards/count_adherence/mean": 0.7916666865348816, "rewards/count_adherence/std": 0.3053750991821289, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9943339228630066, "rewards/repeat_soft/std": 0.013016518205404282, "rewards/judge_quality/mean": 0.25999999046325684, "rewards/judge_quality/std": 0.14774495363235474, "rewards/total_composite/mean": 0.36343780159950256, "rewards/total_composite/std": 0.19955705106258392, "reward": 0.36343780159950256, "reward_std": 0.19955705106258392, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22319409251213074, "sampling/sampling_logp_difference/max": 1.00079345703125, "sampling/importance_sampling_ratio/min": 0.367587685585022, "sampling/importance_sampling_ratio/mean": 1.0507912635803223, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.302706316113472, "clip_ratio/low_mean": 0.031298321671783924, "clip_ratio/low_min": 0.031298321671783924, "clip_ratio/high_mean": 0.13390666618943214, "clip_ratio/high_max": 0.13390666618943214, "clip_ratio/region_mean": 0.16520498786121607, "reward_total_mean": 0.36343780159950256, "reward_meter_mean": 0.19669239223003387, "reward_meter_std": 0.23418445885181427, "reward_count_adherence_mean": 0.7916666865348816, "reward_count_adherence_std": 0.3053750991821289, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9943339228630066, "reward_repeat_soft_std": 0.013016518205404282, "reward_judge_quality_mean": 0.25999999046325684, "reward_judge_quality_std": 0.14774495363235474, "reward_total_composite_mean": 0.36343780159950256, "reward_total_composite_std": 0.19955705106258392} {"timestamp_utc": "2026-04-12T22:50:11Z", "mode": "train", "global_step": 253, "epoch": 0.02541436464088398, "loss": 0.0816, "grad_norm": 15.605833053588867, "learning_rate": 9.236363636363636e-06, "num_tokens": 492673.0, "completions/mean_length": 58.125, "completions/min_length": 37.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.125, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.7679077386856079, "rewards/meter/std": 0.3815578520298004, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9969526529312134, "rewards/repeat_soft/std": 0.004078801721334457, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.2817166745662689, "rewards/total_composite/mean": 0.7445037364959717, "rewards/total_composite/std": 0.21603462100028992, "reward": 0.7445037364959717, "reward_std": 0.21603460609912872, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18630020320415497, "sampling/sampling_logp_difference/max": 1.3970117568969727, "sampling/importance_sampling_ratio/min": 0.24733497202396393, "sampling/importance_sampling_ratio/mean": 1.0316206216812134, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.08673258125782, "clip_ratio/low_mean": 0.07361606135964394, "clip_ratio/low_min": 0.07361606135964394, "clip_ratio/high_mean": 0.12867111526429653, "clip_ratio/high_max": 0.12867111526429653, "clip_ratio/region_mean": 0.20228717662394047, "reward_total_mean": 0.7445037364959717, "reward_meter_mean": 0.7679077386856079, "reward_meter_std": 0.3815578520298004, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9969526529312134, "reward_repeat_soft_std": 0.004078801721334457, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.2817166745662689, "reward_total_composite_mean": 0.7445037364959717, "reward_total_composite_std": 0.21603462100028992} {"timestamp_utc": "2026-04-12T22:50:19Z", "mode": "train", "global_step": 254, "epoch": 0.02551481667503767, "loss": 0.283, "grad_norm": 16.91963005065918, "learning_rate": 9.233333333333334e-06, "num_tokens": 494633.0, "completions/mean_length": 77.0, "completions/min_length": 57.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.0, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.6311284303665161, "rewards/meter/std": 0.3541278541088104, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9968422651290894, "rewards/repeat_soft/std": 0.003317322349175811, "rewards/judge_quality/mean": 0.3399999737739563, "rewards/judge_quality/std": 0.11747340112924576, "rewards/total_composite/mean": 0.5881582498550415, "rewards/total_composite/std": 0.28184789419174194, "reward": 0.5881582498550415, "reward_std": 0.28184786438941956, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2270972579717636, "sampling/sampling_logp_difference/max": 1.8285832405090332, "sampling/importance_sampling_ratio/min": 0.16064099967479706, "sampling/importance_sampling_ratio/mean": 1.0194388628005981, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9485317021608353, "clip_ratio/low_mean": 0.03665293101221323, "clip_ratio/low_min": 0.03665293101221323, "clip_ratio/high_mean": 0.15228915959596634, "clip_ratio/high_max": 0.15228915959596634, "clip_ratio/region_mean": 0.18894209060817957, "reward_total_mean": 0.5881582498550415, "reward_meter_mean": 0.6311284303665161, "reward_meter_std": 0.3541278541088104, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9968422651290894, "reward_repeat_soft_std": 0.003317322349175811, "reward_judge_quality_mean": 0.3399999737739563, "reward_judge_quality_std": 0.11747340112924576, "reward_total_composite_mean": 0.5881582498550415, "reward_total_composite_std": 0.28184789419174194} {"timestamp_utc": "2026-04-12T22:50:26Z", "mode": "train", "global_step": 255, "epoch": 0.025615268709191362, "loss": 0.0933, "grad_norm": 10.187488555908203, "learning_rate": 9.23030303030303e-06, "num_tokens": 496727.0, "completions/mean_length": 87.75, "completions/min_length": 48.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.75, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.8058518171310425, "rewards/meter/std": 0.29976314306259155, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9954677224159241, "rewards/repeat_soft/std": 0.004811787512153387, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.18715444207191467, "rewards/total_composite/mean": 0.6141685247421265, "rewards/total_composite/std": 0.3827317953109741, "reward": 0.6141685247421265, "reward_std": 0.38273176550865173, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18928049504756927, "sampling/sampling_logp_difference/max": 1.6620492935180664, "sampling/importance_sampling_ratio/min": 0.18974971771240234, "sampling/importance_sampling_ratio/mean": 1.0549622774124146, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5076293349266052, "clip_ratio/low_mean": 0.04637789633125067, "clip_ratio/low_min": 0.04637789633125067, "clip_ratio/high_mean": 0.1357775367796421, "clip_ratio/high_max": 0.1357775367796421, "clip_ratio/region_mean": 0.18215543311089277, "reward_total_mean": 0.6141685247421265, "reward_meter_mean": 0.8058518171310425, "reward_meter_std": 0.29976314306259155, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9954677224159241, "reward_repeat_soft_std": 0.004811787512153387, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.18715444207191467, "reward_total_composite_mean": 0.6141685247421265, "reward_total_composite_std": 0.3827317953109741} {"timestamp_utc": "2026-04-12T22:50:34Z", "mode": "train", "global_step": 256, "epoch": 0.025715720743345053, "loss": 0.4695, "grad_norm": 10.185015678405762, "learning_rate": 9.227272727272728e-06, "num_tokens": 498717.0, "completions/mean_length": 75.75, "completions/min_length": 33.0, "completions/max_length": 186.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 186.0, "rewards/meter/mean": 0.6310364007949829, "rewards/meter/std": 0.4851512610912323, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9918019771575928, "rewards/repeat_soft/std": 0.013171611353754997, "rewards/judge_quality/mean": 0.35499998927116394, "rewards/judge_quality/std": 0.20057061314582825, "rewards/total_composite/mean": 0.620896577835083, "rewards/total_composite/std": 0.29314717650413513, "reward": 0.620896577835083, "reward_std": 0.29314714670181274, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21619881689548492, "sampling/sampling_logp_difference/max": 1.594430923461914, "sampling/importance_sampling_ratio/min": 0.20302404463291168, "sampling/importance_sampling_ratio/mean": 1.0597259998321533, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.138188272714615, "clip_ratio/low_mean": 0.0570092536509037, "clip_ratio/low_min": 0.0570092536509037, "clip_ratio/high_mean": 0.13490629941225052, "clip_ratio/high_max": 0.13490629941225052, "clip_ratio/region_mean": 0.19191555306315422, "reward_total_mean": 0.620896577835083, "reward_meter_mean": 0.6310364007949829, "reward_meter_std": 0.4851512610912323, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9918019771575928, "reward_repeat_soft_std": 0.013171611353754997, "reward_judge_quality_mean": 0.35499998927116394, "reward_judge_quality_std": 0.20057061314582825, "reward_total_composite_mean": 0.620896577835083, "reward_total_composite_std": 0.29314717650413513} {"timestamp_utc": "2026-04-12T22:50:45Z", "mode": "train", "global_step": 257, "epoch": 0.025816172777498744, "loss": -0.0738, "grad_norm": 3.497774839401245, "learning_rate": 9.224242424242424e-06, "num_tokens": 501831.0, "completions/mean_length": 240.25, "completions/min_length": 159.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 201.42857360839844, "completions/min_terminated_length": 159.0, "completions/max_terminated_length": 330.0, "rewards/meter/mean": 0.6848487854003906, "rewards/meter/std": 0.38694190979003906, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.2618614733219147, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9968925714492798, "rewards/repeat_soft/std": 0.004322752822190523, "rewards/judge_quality/mean": 0.22999998927116394, "rewards/judge_quality/std": 0.1086278036236763, "rewards/total_composite/mean": 0.5624604225158691, "rewards/total_composite/std": 0.29574236273765564, "reward": 0.5624604225158691, "reward_std": 0.29574233293533325, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20623661577701569, "sampling/sampling_logp_difference/max": 1.4393281936645508, "sampling/importance_sampling_ratio/min": 0.23708698153495789, "sampling/importance_sampling_ratio/mean": 1.0393543243408203, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.9671258628368378, "clip_ratio/low_mean": 0.01401515118777752, "clip_ratio/low_min": 0.01401515118777752, "clip_ratio/high_mean": 0.14178386144340038, "clip_ratio/high_max": 0.14178386144340038, "clip_ratio/region_mean": 0.1557990126311779, "reward_total_mean": 0.5624604225158691, "reward_meter_mean": 0.6848487854003906, "reward_meter_std": 0.38694190979003906, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.2618614733219147, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9968925714492798, "reward_repeat_soft_std": 0.004322752822190523, "reward_judge_quality_mean": 0.22999998927116394, "reward_judge_quality_std": 0.1086278036236763, "reward_total_composite_mean": 0.5624604225158691, "reward_total_composite_std": 0.29574236273765564} {"timestamp_utc": "2026-04-12T22:50:55Z", "mode": "train", "global_step": 258, "epoch": 0.025916624811652435, "loss": 0.398, "grad_norm": 8.336535453796387, "learning_rate": 9.221212121212123e-06, "num_tokens": 504243.0, "completions/mean_length": 142.5, "completions/min_length": 83.0, "completions/max_length": 342.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.5, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 342.0, "rewards/meter/mean": 0.30017322301864624, "rewards/meter/std": 0.35891056060791016, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.18898223340511322, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9986302852630615, "rewards/repeat_soft/std": 0.0016452295240014791, "rewards/judge_quality/mean": 0.42625001072883606, "rewards/judge_quality/std": 0.2661866843700409, "rewards/total_composite/mean": 0.47006696462631226, "rewards/total_composite/std": 0.2600260376930237, "reward": 0.47006696462631226, "reward_std": 0.2600260376930237, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21126966178417206, "sampling/sampling_logp_difference/max": 1.8719696998596191, "sampling/importance_sampling_ratio/min": 0.15382038056850433, "sampling/importance_sampling_ratio/mean": 1.0416011810302734, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.09677854180336, "clip_ratio/low_mean": 0.07471582293510437, "clip_ratio/low_min": 0.07471582293510437, "clip_ratio/high_mean": 0.11672787554562092, "clip_ratio/high_max": 0.11672787554562092, "clip_ratio/region_mean": 0.1914436984807253, "reward_total_mean": 0.47006696462631226, "reward_meter_mean": 0.30017322301864624, "reward_meter_std": 0.35891056060791016, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.18898223340511322, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9986302852630615, "reward_repeat_soft_std": 0.0016452295240014791, "reward_judge_quality_mean": 0.42625001072883606, "reward_judge_quality_std": 0.2661866843700409, "reward_total_composite_mean": 0.47006696462631226, "reward_total_composite_std": 0.2600260376930237} {"timestamp_utc": "2026-04-12T22:51:06Z", "mode": "train", "global_step": 259, "epoch": 0.026017076845806127, "loss": -0.0828, "grad_norm": 3.6295735836029053, "learning_rate": 9.21818181818182e-06, "num_tokens": 505682.0, "completions/mean_length": 87.875, "completions/min_length": 18.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 27.285715103149414, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.6480820178985596, "rewards/meter/std": 0.47923460602760315, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9576854109764099, "rewards/repeat_soft/std": 0.01361766830086708, "rewards/judge_quality/mean": 0.3050000071525574, "rewards/judge_quality/std": 0.15287716686725616, "rewards/total_composite/mean": 0.5966835021972656, "rewards/total_composite/std": 0.3035220205783844, "reward": 0.5966835021972656, "reward_std": 0.3035220205783844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19513347744941711, "sampling/sampling_logp_difference/max": 0.9347000122070312, "sampling/importance_sampling_ratio/min": 0.392703652381897, "sampling/importance_sampling_ratio/mean": 1.0420039892196655, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9209102094173431, "clip_ratio/low_mean": 0.057539683766663074, "clip_ratio/low_min": 0.057539683766663074, "clip_ratio/high_mean": 0.11660980340093374, "clip_ratio/high_max": 0.11660980340093374, "clip_ratio/region_mean": 0.17414948716759682, "reward_total_mean": 0.5966835021972656, "reward_meter_mean": 0.6480820178985596, "reward_meter_std": 0.47923460602760315, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9576854109764099, "reward_repeat_soft_std": 0.01361766830086708, "reward_judge_quality_mean": 0.3050000071525574, "reward_judge_quality_std": 0.15287716686725616, "reward_total_composite_mean": 0.5966835021972656, "reward_total_composite_std": 0.3035220205783844} {"timestamp_utc": "2026-04-12T22:51:14Z", "mode": "train", "global_step": 260, "epoch": 0.026117528879959818, "loss": 0.0586, "grad_norm": 13.4459810256958, "learning_rate": 9.215151515151515e-06, "num_tokens": 507298.0, "completions/mean_length": 46.0, "completions/min_length": 41.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.0, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.665064811706543, "rewards/meter/std": 0.35738658905029297, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9975134134292603, "rewards/repeat_soft/std": 0.005810212343931198, "rewards/judge_quality/mean": 0.5774999856948853, "rewards/judge_quality/std": 0.2988908290863037, "rewards/total_composite/mean": 0.7222805023193359, "rewards/total_composite/std": 0.2268882691860199, "reward": 0.7222805023193359, "reward_std": 0.2268882691860199, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20865072309970856, "sampling/sampling_logp_difference/max": 1.286428451538086, "sampling/importance_sampling_ratio/min": 0.27625566720962524, "sampling/importance_sampling_ratio/mean": 1.028465986251831, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2926181703805923, "clip_ratio/low_mean": 0.10158740729093552, "clip_ratio/low_min": 0.10158740729093552, "clip_ratio/high_mean": 0.09091721381992102, "clip_ratio/high_max": 0.09091721381992102, "clip_ratio/region_mean": 0.19250462111085653, "reward_total_mean": 0.7222805023193359, "reward_meter_mean": 0.665064811706543, "reward_meter_std": 0.35738658905029297, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9975134134292603, "reward_repeat_soft_std": 0.005810212343931198, "reward_judge_quality_mean": 0.5774999856948853, "reward_judge_quality_std": 0.2988908290863037, "reward_total_composite_mean": 0.7222805023193359, "reward_total_composite_std": 0.2268882691860199} {"timestamp_utc": "2026-04-12T22:51:21Z", "mode": "train", "global_step": 261, "epoch": 0.026217980914113512, "loss": 0.0119, "grad_norm": 10.644651412963867, "learning_rate": 9.212121212121213e-06, "num_tokens": 509068.0, "completions/mean_length": 63.25, "completions/min_length": 56.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.25, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.5956026911735535, "rewards/meter/std": 0.3906526565551758, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9973117113113403, "rewards/repeat_soft/std": 0.0033813740592449903, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.6385023593902588, "rewards/total_composite/std": 0.17080283164978027, "reward": 0.6385023593902588, "reward_std": 0.17080283164978027, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19046685099601746, "sampling/sampling_logp_difference/max": 1.0633344650268555, "sampling/importance_sampling_ratio/min": 0.3453024923801422, "sampling/importance_sampling_ratio/mean": 1.0462766885757446, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6052364706993103, "clip_ratio/low_mean": 0.06559686735272408, "clip_ratio/low_min": 0.06559686735272408, "clip_ratio/high_mean": 0.09455354698002338, "clip_ratio/high_max": 0.09455354698002338, "clip_ratio/region_mean": 0.16015041433274746, "reward_total_mean": 0.6385023593902588, "reward_meter_mean": 0.5956026911735535, "reward_meter_std": 0.3906526565551758, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9973117113113403, "reward_repeat_soft_std": 0.0033813740592449903, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.6385023593902588, "reward_total_composite_std": 0.17080283164978027} {"timestamp_utc": "2026-04-12T22:51:33Z", "mode": "train", "global_step": 262, "epoch": 0.026318432948267204, "loss": -0.0573, "grad_norm": 2.3920626640319824, "learning_rate": 9.20909090909091e-06, "num_tokens": 510579.0, "completions/mean_length": 225.875, "completions/min_length": 42.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 54.20000076293945, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.3869282305240631, "rewards/meter/std": 0.49040281772613525, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.26726123690605164, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9937890768051147, "rewards/repeat_soft/std": 0.012971606105566025, "rewards/judge_quality/mean": 0.20499998331069946, "rewards/judge_quality/std": 0.15657037496566772, "rewards/total_composite/mean": 0.31038329005241394, "rewards/total_composite/std": 0.330401748418808, "reward": 0.31038329005241394, "reward_std": 0.3304017186164856, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22076906263828278, "sampling/sampling_logp_difference/max": 1.3914079666137695, "sampling/importance_sampling_ratio/min": 0.24872486293315887, "sampling/importance_sampling_ratio/mean": 1.031574010848999, "sampling/importance_sampling_ratio/max": 1.9119187593460083, "entropy": 1.8284253776073456, "clip_ratio/low_mean": 0.02330508455634117, "clip_ratio/low_min": 0.02330508455634117, "clip_ratio/high_mean": 0.08193016704171896, "clip_ratio/high_max": 0.08193016704171896, "clip_ratio/region_mean": 0.10523525159806013, "reward_total_mean": 0.31038329005241394, "reward_meter_mean": 0.3869282305240631, "reward_meter_std": 0.49040281772613525, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.26726123690605164, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9937890768051147, "reward_repeat_soft_std": 0.012971606105566025, "reward_judge_quality_mean": 0.20499998331069946, "reward_judge_quality_std": 0.15657037496566772, "reward_total_composite_mean": 0.31038329005241394, "reward_total_composite_std": 0.330401748418808} {"timestamp_utc": "2026-04-12T22:51:45Z", "mode": "train", "global_step": 263, "epoch": 0.026418884982420895, "loss": -0.0996, "grad_norm": 4.593531131744385, "learning_rate": 9.206060606060607e-06, "num_tokens": 512677.0, "completions/mean_length": 154.25, "completions/min_length": 78.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 103.14286041259766, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.3967878222465515, "rewards/meter/std": 0.37048086524009705, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.26726123690605164, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9905542731285095, "rewards/repeat_soft/std": 0.012595550157129765, "rewards/judge_quality/mean": 0.3187499940395355, "rewards/judge_quality/std": 0.14961259067058563, "rewards/total_composite/mean": 0.43280234932899475, "rewards/total_composite/std": 0.30939069390296936, "reward": 0.43280234932899475, "reward_std": 0.309390664100647, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21722321212291718, "sampling/sampling_logp_difference/max": 1.5441341400146484, "sampling/importance_sampling_ratio/min": 0.21349665522575378, "sampling/importance_sampling_ratio/mean": 1.0531666278839111, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.69633412361145, "clip_ratio/low_mean": 0.05390256177634001, "clip_ratio/low_min": 0.05390256177634001, "clip_ratio/high_mean": 0.097511762753129, "clip_ratio/high_max": 0.097511762753129, "clip_ratio/region_mean": 0.151414324529469, "reward_total_mean": 0.43280234932899475, "reward_meter_mean": 0.3967878222465515, "reward_meter_std": 0.37048086524009705, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.26726123690605164, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9905542731285095, "reward_repeat_soft_std": 0.012595550157129765, "reward_judge_quality_mean": 0.3187499940395355, "reward_judge_quality_std": 0.14961259067058563, "reward_total_composite_mean": 0.43280234932899475, "reward_total_composite_std": 0.30939069390296936} {"timestamp_utc": "2026-04-12T22:51:57Z", "mode": "train", "global_step": 264, "epoch": 0.026519337016574586, "loss": -0.1188, "grad_norm": 2.7556891441345215, "learning_rate": 9.203030303030304e-06, "num_tokens": 514642.0, "completions/mean_length": 262.625, "completions/min_length": 75.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 113.0, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.5223958492279053, "rewards/meter/std": 0.3973325192928314, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.18600596487522125, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9953274726867676, "rewards/repeat_soft/std": 0.0034347306936979294, "rewards/judge_quality/mean": 0.2199999988079071, "rewards/judge_quality/std": 0.18205179274082184, "rewards/total_composite/mean": 0.3639429211616516, "rewards/total_composite/std": 0.35092559456825256, "reward": 0.3639429211616516, "reward_std": 0.3509255647659302, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18799443542957306, "sampling/sampling_logp_difference/max": 1.7502002716064453, "sampling/importance_sampling_ratio/min": 0.17373913526535034, "sampling/importance_sampling_ratio/mean": 1.0437512397766113, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5592512637376785, "clip_ratio/low_mean": 0.03595762327313423, "clip_ratio/low_min": 0.03595762327313423, "clip_ratio/high_mean": 0.07350747846066952, "clip_ratio/high_max": 0.07350747846066952, "clip_ratio/region_mean": 0.10946510173380375, "reward_total_mean": 0.3639429211616516, "reward_meter_mean": 0.5223958492279053, "reward_meter_std": 0.3973325192928314, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.18600596487522125, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9953274726867676, "reward_repeat_soft_std": 0.0034347306936979294, "reward_judge_quality_mean": 0.2199999988079071, "reward_judge_quality_std": 0.18205179274082184, "reward_total_composite_mean": 0.3639429211616516, "reward_total_composite_std": 0.35092559456825256} {"timestamp_utc": "2026-04-12T22:52:05Z", "mode": "train", "global_step": 265, "epoch": 0.026619789050728277, "loss": 0.0676, "grad_norm": 13.086560249328613, "learning_rate": 9.200000000000002e-06, "num_tokens": 516837.0, "completions/mean_length": 97.375, "completions/min_length": 69.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.375, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.5505688190460205, "rewards/meter/std": 0.34225714206695557, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9939977526664734, "rewards/repeat_soft/std": 0.005710379686206579, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.6242807507514954, "rewards/total_composite/std": 0.18888473510742188, "reward": 0.6242807507514954, "reward_std": 0.18888473510742188, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2172665297985077, "sampling/sampling_logp_difference/max": 1.6288820505142212, "sampling/importance_sampling_ratio/min": 0.19614872336387634, "sampling/importance_sampling_ratio/mean": 1.0174729824066162, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.889648199081421, "clip_ratio/low_mean": 0.06688261404633522, "clip_ratio/low_min": 0.06688261404633522, "clip_ratio/high_mean": 0.1411463413387537, "clip_ratio/high_max": 0.1411463413387537, "clip_ratio/region_mean": 0.20802895538508892, "reward_total_mean": 0.6242807507514954, "reward_meter_mean": 0.5505688190460205, "reward_meter_std": 0.34225714206695557, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9939977526664734, "reward_repeat_soft_std": 0.005710379686206579, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.6242807507514954, "reward_total_composite_std": 0.18888473510742188} {"timestamp_utc": "2026-04-12T22:52:11Z", "mode": "train", "global_step": 266, "epoch": 0.026720241084881968, "loss": -0.005, "grad_norm": 15.39251708984375, "learning_rate": 9.196969696969697e-06, "num_tokens": 518236.0, "completions/mean_length": 24.875, "completions/min_length": 19.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.875, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9473345279693604, "rewards/meter/std": 0.09497682005167007, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.5087499618530273, "rewards/judge_quality/std": 0.16617010533809662, "rewards/total_composite/mean": 0.8251755237579346, "rewards/total_composite/std": 0.0686953216791153, "reward": 0.8251755237579346, "reward_std": 0.0686952993273735, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20876620709896088, "sampling/sampling_logp_difference/max": 1.2854156494140625, "sampling/importance_sampling_ratio/min": 0.27653560042381287, "sampling/importance_sampling_ratio/mean": 1.0348933935165405, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.1556093096733093, "clip_ratio/low_mean": 0.14158289786428213, "clip_ratio/low_min": 0.14158289786428213, "clip_ratio/high_mean": 0.08943452406674623, "clip_ratio/high_max": 0.08943452406674623, "clip_ratio/region_mean": 0.23101742193102837, "reward_total_mean": 0.8251755237579346, "reward_meter_mean": 0.9473345279693604, "reward_meter_std": 0.09497682005167007, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.5087499618530273, "reward_judge_quality_std": 0.16617010533809662, "reward_total_composite_mean": 0.8251755237579346, "reward_total_composite_std": 0.0686953216791153} {"timestamp_utc": "2026-04-12T22:52:23Z", "mode": "train", "global_step": 267, "epoch": 0.02682069311903566, "loss": -0.061, "grad_norm": 2.8569726943969727, "learning_rate": 9.193939393939395e-06, "num_tokens": 519867.0, "completions/mean_length": 171.875, "completions/min_length": 41.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 58.5, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.7095774412155151, "rewards/meter/std": 0.34361889958381653, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.25877460837364197, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9946221709251404, "rewards/repeat_soft/std": 0.013042791746556759, "rewards/judge_quality/mean": 0.2512499988079071, "rewards/judge_quality/std": 0.15887437760829926, "rewards/total_composite/mean": 0.4705522656440735, "rewards/total_composite/std": 0.3202861249446869, "reward": 0.4705522656440735, "reward_std": 0.3202861249446869, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22520560026168823, "sampling/sampling_logp_difference/max": 1.5406584739685059, "sampling/importance_sampling_ratio/min": 0.21423998475074768, "sampling/importance_sampling_ratio/mean": 1.0602422952651978, "sampling/importance_sampling_ratio/max": 1.930098533630371, "entropy": 2.7558106184005737, "clip_ratio/low_mean": 0.05927700363099575, "clip_ratio/low_min": 0.05927700363099575, "clip_ratio/high_mean": 0.08057818748056889, "clip_ratio/high_max": 0.08057818748056889, "clip_ratio/region_mean": 0.13985519111156464, "reward_total_mean": 0.4705522656440735, "reward_meter_mean": 0.7095774412155151, "reward_meter_std": 0.34361889958381653, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.25877460837364197, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9946221709251404, "reward_repeat_soft_std": 0.013042791746556759, "reward_judge_quality_mean": 0.2512499988079071, "reward_judge_quality_std": 0.15887437760829926, "reward_total_composite_mean": 0.4705522656440735, "reward_total_composite_std": 0.3202861249446869} {"timestamp_utc": "2026-04-12T22:52:30Z", "mode": "train", "global_step": 268, "epoch": 0.02692114515318935, "loss": 0.2599, "grad_norm": 25.378992080688477, "learning_rate": 9.190909090909092e-06, "num_tokens": 521448.0, "completions/mean_length": 30.625, "completions/min_length": 25.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.625, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.7664294242858887, "rewards/meter/std": 0.3176596760749817, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9845855236053467, "rewards/repeat_soft/std": 0.007975445128977299, "rewards/judge_quality/mean": 0.8362500667572021, "rewards/judge_quality/std": 0.23688077926635742, "rewards/total_composite/mean": 0.8442267179489136, "rewards/total_composite/std": 0.21112513542175293, "reward": 0.8442267179489136, "reward_std": 0.21112512052059174, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11083001643419266, "sampling/sampling_logp_difference/max": 2.0658926963806152, "sampling/importance_sampling_ratio/min": 0.1267051249742508, "sampling/importance_sampling_ratio/mean": 1.0101468563079834, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5642402805387974, "clip_ratio/low_mean": 0.014999999664723873, "clip_ratio/low_min": 0.014999999664723873, "clip_ratio/high_mean": 0.03855820186436176, "clip_ratio/high_max": 0.03855820186436176, "clip_ratio/region_mean": 0.053558201529085636, "reward_total_mean": 0.8442267179489136, "reward_meter_mean": 0.7664294242858887, "reward_meter_std": 0.3176596760749817, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9845855236053467, "reward_repeat_soft_std": 0.007975445128977299, "reward_judge_quality_mean": 0.8362500667572021, "reward_judge_quality_std": 0.23688077926635742, "reward_total_composite_mean": 0.8442267179489136, "reward_total_composite_std": 0.21112513542175293} {"timestamp_utc": "2026-04-12T22:52:37Z", "mode": "train", "global_step": 269, "epoch": 0.027021597187343045, "loss": 0.0385, "grad_norm": 16.914997100830078, "learning_rate": 9.187878787878789e-06, "num_tokens": 522883.0, "completions/mean_length": 29.375, "completions/min_length": 25.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.375, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9626954793930054, "rewards/meter/std": 0.05454079061746597, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9612975120544434, "rewards/repeat_soft/std": 0.0034009767696261406, "rewards/judge_quality/mean": 0.4762499928474426, "rewards/judge_quality/std": 0.19167961180210114, "rewards/total_composite/mean": 0.8222177028656006, "rewards/total_composite/std": 0.06246612221002579, "reward": 0.8222177028656006, "reward_std": 0.06246611848473549, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2157377451658249, "sampling/sampling_logp_difference/max": 3.440803050994873, "sampling/importance_sampling_ratio/min": 0.03203894570469856, "sampling/importance_sampling_ratio/mean": 1.0384995937347412, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.321295276284218, "clip_ratio/low_mean": 0.1274358108639717, "clip_ratio/low_min": 0.1274358108639717, "clip_ratio/high_mean": 0.045026881620287895, "clip_ratio/high_max": 0.045026881620287895, "clip_ratio/region_mean": 0.1724626924842596, "reward_total_mean": 0.8222177028656006, "reward_meter_mean": 0.9626954793930054, "reward_meter_std": 0.05454079061746597, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9612975120544434, "reward_repeat_soft_std": 0.0034009767696261406, "reward_judge_quality_mean": 0.4762499928474426, "reward_judge_quality_std": 0.19167961180210114, "reward_total_composite_mean": 0.8222177028656006, "reward_total_composite_std": 0.06246612221002579} {"timestamp_utc": "2026-04-12T22:52:48Z", "mode": "train", "global_step": 270, "epoch": 0.027122049221496736, "loss": -0.1131, "grad_norm": 3.7887232303619385, "learning_rate": 9.184848484848485e-06, "num_tokens": 524530.0, "completions/mean_length": 114.875, "completions/min_length": 38.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 58.142860412597656, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.5569263696670532, "rewards/meter/std": 0.4289514124393463, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9899011850357056, "rewards/repeat_soft/std": 0.013745320960879326, "rewards/judge_quality/mean": 0.3174999952316284, "rewards/judge_quality/std": 0.14210157096385956, "rewards/total_composite/mean": 0.5180926322937012, "rewards/total_composite/std": 0.2810794711112976, "reward": 0.5180926322937012, "reward_std": 0.2810794711112976, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20393048226833344, "sampling/sampling_logp_difference/max": 1.0139942169189453, "sampling/importance_sampling_ratio/min": 0.36276713013648987, "sampling/importance_sampling_ratio/mean": 1.0665736198425293, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.9601809084415436, "clip_ratio/low_mean": 0.05279624182730913, "clip_ratio/low_min": 0.05279624182730913, "clip_ratio/high_mean": 0.0872265063226223, "clip_ratio/high_max": 0.0872265063226223, "clip_ratio/region_mean": 0.14002274814993143, "reward_total_mean": 0.5180926322937012, "reward_meter_mean": 0.5569263696670532, "reward_meter_std": 0.4289514124393463, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9899011850357056, "reward_repeat_soft_std": 0.013745320960879326, "reward_judge_quality_mean": 0.3174999952316284, "reward_judge_quality_std": 0.14210157096385956, "reward_total_composite_mean": 0.5180926322937012, "reward_total_composite_std": 0.2810794711112976} {"timestamp_utc": "2026-04-12T22:52:59Z", "mode": "train", "global_step": 271, "epoch": 0.027222501255650428, "loss": -0.1001, "grad_norm": 2.131558656692505, "learning_rate": 9.181818181818184e-06, "num_tokens": 526066.0, "completions/mean_length": 160.0, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 42.66666793823242, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.5715053081512451, "rewards/meter/std": 0.36980071663856506, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.25877460837364197, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9958885908126831, "rewards/repeat_soft/std": 0.005992755759507418, "rewards/judge_quality/mean": 0.2549999952316284, "rewards/judge_quality/std": 0.16370704770088196, "rewards/total_composite/mean": 0.4784981608390808, "rewards/total_composite/std": 0.26447972655296326, "reward": 0.4784981608390808, "reward_std": 0.26447972655296326, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2198251187801361, "sampling/sampling_logp_difference/max": 1.063920021057129, "sampling/importance_sampling_ratio/min": 0.3451003432273865, "sampling/importance_sampling_ratio/mean": 1.0834839344024658, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5456457436084747, "clip_ratio/low_mean": 0.03544494789093733, "clip_ratio/low_min": 0.03544494789093733, "clip_ratio/high_mean": 0.11276789102703333, "clip_ratio/high_max": 0.11276789102703333, "clip_ratio/region_mean": 0.14821283891797066, "reward_total_mean": 0.4784981608390808, "reward_meter_mean": 0.5715053081512451, "reward_meter_std": 0.36980071663856506, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.25877460837364197, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9958885908126831, "reward_repeat_soft_std": 0.005992755759507418, "reward_judge_quality_mean": 0.2549999952316284, "reward_judge_quality_std": 0.16370704770088196, "reward_total_composite_mean": 0.4784981608390808, "reward_total_composite_std": 0.26447972655296326} {"timestamp_utc": "2026-04-12T22:53:08Z", "mode": "train", "global_step": 272, "epoch": 0.02732295328980412, "loss": 0.773, "grad_norm": 13.107495307922363, "learning_rate": 9.178787878787879e-06, "num_tokens": 527994.0, "completions/mean_length": 84.0, "completions/min_length": 36.0, "completions/max_length": 252.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.0, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 252.0, "rewards/meter/mean": 0.938515841960907, "rewards/meter/std": 0.09587184339761734, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9938917756080627, "rewards/repeat_soft/std": 0.008453421294689178, "rewards/judge_quality/mean": 0.42250001430511475, "rewards/judge_quality/std": 0.24364787340164185, "rewards/total_composite/mean": 0.7130944728851318, "rewards/total_composite/std": 0.29630953073501587, "reward": 0.7130944728851318, "reward_std": 0.2963095009326935, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2045411467552185, "sampling/sampling_logp_difference/max": 1.827183723449707, "sampling/importance_sampling_ratio/min": 0.16086597740650177, "sampling/importance_sampling_ratio/mean": 1.0493766069412231, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.327467992901802, "clip_ratio/low_mean": 0.027777778450399637, "clip_ratio/low_min": 0.027777778450399637, "clip_ratio/high_mean": 0.11033590789884329, "clip_ratio/high_max": 0.11033590789884329, "clip_ratio/region_mean": 0.13811368634924293, "reward_total_mean": 0.7130944728851318, "reward_meter_mean": 0.938515841960907, "reward_meter_std": 0.09587184339761734, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9938917756080627, "reward_repeat_soft_std": 0.008453421294689178, "reward_judge_quality_mean": 0.42250001430511475, "reward_judge_quality_std": 0.24364787340164185, "reward_total_composite_mean": 0.7130944728851318, "reward_total_composite_std": 0.29630953073501587} {"timestamp_utc": "2026-04-12T22:53:19Z", "mode": "train", "global_step": 273, "epoch": 0.02742340532395781, "loss": 0.1127, "grad_norm": 3.209653854370117, "learning_rate": 9.175757575757576e-06, "num_tokens": 530190.0, "completions/mean_length": 236.5, "completions/min_length": 76.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 144.6666717529297, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 451.0, "rewards/meter/mean": 0.25925591588020325, "rewards/meter/std": 0.33102232217788696, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.5, "rewards/hard_gate/std": 0.5345224738121033, "rewards/repeat_soft/mean": 0.9986441731452942, "rewards/repeat_soft/std": 0.0013439194299280643, "rewards/judge_quality/mean": 0.2762500047683716, "rewards/judge_quality/std": 0.23700135946273804, "rewards/total_composite/mean": 0.2862798571586609, "rewards/total_composite/std": 0.3258228003978729, "reward": 0.2862798571586609, "reward_std": 0.32582277059555054, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2007029950618744, "sampling/sampling_logp_difference/max": 1.234074592590332, "sampling/importance_sampling_ratio/min": 0.33914536237716675, "sampling/importance_sampling_ratio/mean": 1.0485591888427734, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.158889263868332, "clip_ratio/low_mean": 0.02876869961619377, "clip_ratio/low_min": 0.02876869961619377, "clip_ratio/high_mean": 0.06529024709016085, "clip_ratio/high_max": 0.06529024709016085, "clip_ratio/region_mean": 0.09405894670635462, "reward_total_mean": 0.2862798571586609, "reward_meter_mean": 0.25925591588020325, "reward_meter_std": 0.33102232217788696, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.5, "reward_hard_gate_std": 0.5345224738121033, "reward_repeat_soft_mean": 0.9986441731452942, "reward_repeat_soft_std": 0.0013439194299280643, "reward_judge_quality_mean": 0.2762500047683716, "reward_judge_quality_std": 0.23700135946273804, "reward_total_composite_mean": 0.2862798571586609, "reward_total_composite_std": 0.3258228003978729} {"timestamp_utc": "2026-04-12T22:53:26Z", "mode": "train", "global_step": 274, "epoch": 0.0275238573581115, "loss": 0.0062, "grad_norm": 23.61359977722168, "learning_rate": 9.172727272727274e-06, "num_tokens": 531596.0, "completions/mean_length": 22.75, "completions/min_length": 18.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.75, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.7578386068344116, "rewards/meter/std": 0.385356068611145, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.36374998092651367, "rewards/judge_quality/std": 0.09500939399003983, "rewards/total_composite/mean": 0.696402370929718, "rewards/total_composite/std": 0.19662360846996307, "reward": 0.696402370929718, "reward_std": 0.19662362337112427, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2527177035808563, "sampling/sampling_logp_difference/max": 1.722092628479004, "sampling/importance_sampling_ratio/min": 0.1786918193101883, "sampling/importance_sampling_ratio/mean": 1.0275429487228394, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6737553775310516, "clip_ratio/low_mean": 0.04503105580806732, "clip_ratio/low_min": 0.04503105580806732, "clip_ratio/high_mean": 0.14321525860577822, "clip_ratio/high_max": 0.14321525860577822, "clip_ratio/region_mean": 0.18824631441384554, "reward_total_mean": 0.696402370929718, "reward_meter_mean": 0.7578386068344116, "reward_meter_std": 0.385356068611145, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.36374998092651367, "reward_judge_quality_std": 0.09500939399003983, "reward_total_composite_mean": 0.696402370929718, "reward_total_composite_std": 0.19662360846996307} {"timestamp_utc": "2026-04-12T22:53:37Z", "mode": "train", "global_step": 275, "epoch": 0.027624309392265192, "loss": -0.1405, "grad_norm": 3.8702547550201416, "learning_rate": 9.169696969696971e-06, "num_tokens": 533602.0, "completions/mean_length": 143.75, "completions/min_length": 67.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 91.14286041259766, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.5579936504364014, "rewards/meter/std": 0.3729588985443115, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.18600596487522125, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9940185546875, "rewards/repeat_soft/std": 0.009340805001556873, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.13715866208076477, "rewards/total_composite/mean": 0.5516610145568848, "rewards/total_composite/std": 0.2865002453327179, "reward": 0.5516610145568848, "reward_std": 0.2865002453327179, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20496076345443726, "sampling/sampling_logp_difference/max": 1.5480213165283203, "sampling/importance_sampling_ratio/min": 0.21266835927963257, "sampling/importance_sampling_ratio/mean": 1.0460983514785767, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.8480335474014282, "clip_ratio/low_mean": 0.07093286514282227, "clip_ratio/low_min": 0.07093286514282227, "clip_ratio/high_mean": 0.11970594339072704, "clip_ratio/high_max": 0.11970594339072704, "clip_ratio/region_mean": 0.1906388085335493, "reward_total_mean": 0.5516610145568848, "reward_meter_mean": 0.5579936504364014, "reward_meter_std": 0.3729588985443115, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.18600596487522125, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9940185546875, "reward_repeat_soft_std": 0.009340805001556873, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.13715866208076477, "reward_total_composite_mean": 0.5516610145568848, "reward_total_composite_std": 0.2865002453327179} {"timestamp_utc": "2026-04-12T22:53:49Z", "mode": "train", "global_step": 276, "epoch": 0.027724761426418883, "loss": -0.0953, "grad_norm": 4.172479152679443, "learning_rate": 9.166666666666666e-06, "num_tokens": 535234.0, "completions/mean_length": 108.0, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 50.28571701049805, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.6255463361740112, "rewards/meter/std": 0.4673767387866974, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9741201400756836, "rewards/repeat_soft/std": 0.037372831255197525, "rewards/judge_quality/mean": 0.5149999856948853, "rewards/judge_quality/std": 0.27244922518730164, "rewards/total_composite/mean": 0.5519406795501709, "rewards/total_composite/std": 0.37978243827819824, "reward": 0.5519406795501709, "reward_std": 0.37978240847587585, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1883494257926941, "sampling/sampling_logp_difference/max": 1.2578630447387695, "sampling/importance_sampling_ratio/min": 0.2842608094215393, "sampling/importance_sampling_ratio/mean": 1.0269664525985718, "sampling/importance_sampling_ratio/max": 1.897179365158081, "entropy": 1.9125686287879944, "clip_ratio/low_mean": 0.0620159599930048, "clip_ratio/low_min": 0.0620159599930048, "clip_ratio/high_mean": 0.07864534854888916, "clip_ratio/high_max": 0.07864534854888916, "clip_ratio/region_mean": 0.14066130854189396, "reward_total_mean": 0.5519406795501709, "reward_meter_mean": 0.6255463361740112, "reward_meter_std": 0.4673767387866974, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9741201400756836, "reward_repeat_soft_std": 0.037372831255197525, "reward_judge_quality_mean": 0.5149999856948853, "reward_judge_quality_std": 0.27244922518730164, "reward_total_composite_mean": 0.5519406795501709, "reward_total_composite_std": 0.37978243827819824} {"timestamp_utc": "2026-04-12T22:53:55Z", "mode": "train", "global_step": 277, "epoch": 0.027825213460572578, "loss": -0.0771, "grad_norm": 15.541146278381348, "learning_rate": 9.163636363636365e-06, "num_tokens": 536741.0, "completions/mean_length": 42.375, "completions/min_length": 34.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.375, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.5097652673721313, "rewards/meter/std": 0.3965023458003998, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9976564049720764, "rewards/repeat_soft/std": 0.0031486700754612684, "rewards/judge_quality/mean": 0.5900000333786011, "rewards/judge_quality/std": 0.22696760296821594, "rewards/total_composite/mean": 0.5524407029151917, "rewards/total_composite/std": 0.2912794351577759, "reward": 0.5524407029151917, "reward_std": 0.2912794351577759, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1873490959405899, "sampling/sampling_logp_difference/max": 1.4329261779785156, "sampling/importance_sampling_ratio/min": 0.2386096864938736, "sampling/importance_sampling_ratio/mean": 1.0438122749328613, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.884998619556427, "clip_ratio/low_mean": 0.10560483857989311, "clip_ratio/low_min": 0.10560483857989311, "clip_ratio/high_mean": 0.09432234615087509, "clip_ratio/high_max": 0.09432234615087509, "clip_ratio/region_mean": 0.1999271847307682, "reward_total_mean": 0.5524407029151917, "reward_meter_mean": 0.5097652673721313, "reward_meter_std": 0.3965023458003998, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9976564049720764, "reward_repeat_soft_std": 0.0031486700754612684, "reward_judge_quality_mean": 0.5900000333786011, "reward_judge_quality_std": 0.22696760296821594, "reward_total_composite_mean": 0.5524407029151917, "reward_total_composite_std": 0.2912794351577759} {"timestamp_utc": "2026-04-12T22:54:01Z", "mode": "train", "global_step": 278, "epoch": 0.02792566549472627, "loss": 0.2152, "grad_norm": 35.80637741088867, "learning_rate": 9.160606060606061e-06, "num_tokens": 538226.0, "completions/mean_length": 32.625, "completions/min_length": 28.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.625, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.8898208141326904, "rewards/meter/std": 0.2734966576099396, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9793791174888611, "rewards/repeat_soft/std": 0.021093856543302536, "rewards/judge_quality/mean": 0.9200000166893005, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.9243572950363159, "rewards/total_composite/std": 0.12395155429840088, "reward": 0.9243572950363159, "reward_std": 0.12395156174898148, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18469545245170593, "sampling/sampling_logp_difference/max": 2.2828831672668457, "sampling/importance_sampling_ratio/min": 0.1019897311925888, "sampling/importance_sampling_ratio/mean": 1.0048621892929077, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8186254575848579, "clip_ratio/low_mean": 0.017045455053448677, "clip_ratio/low_min": 0.017045455053448677, "clip_ratio/high_mean": 0.14566979417577386, "clip_ratio/high_max": 0.14566979417577386, "clip_ratio/region_mean": 0.16271524922922254, "reward_total_mean": 0.9243572950363159, "reward_meter_mean": 0.8898208141326904, "reward_meter_std": 0.2734966576099396, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9793791174888611, "reward_repeat_soft_std": 0.021093856543302536, "reward_judge_quality_mean": 0.9200000166893005, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.9243572950363159, "reward_total_composite_std": 0.12395155429840088} {"timestamp_utc": "2026-04-12T22:54:07Z", "mode": "train", "global_step": 279, "epoch": 0.02802611752887996, "loss": -0.0473, "grad_norm": 10.809500694274902, "learning_rate": 9.157575757575758e-06, "num_tokens": 539967.0, "completions/mean_length": 62.625, "completions/min_length": 51.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.625, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9702678918838501, "rewards/meter/std": 0.07130211591720581, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.989289402961731, "rewards/repeat_soft/std": 0.013245010748505592, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.24833375215530396, "rewards/total_composite/mean": 0.8355494737625122, "rewards/total_composite/std": 0.1138211116194725, "reward": 0.8355494737625122, "reward_std": 0.1138211041688919, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17409032583236694, "sampling/sampling_logp_difference/max": 1.400996208190918, "sampling/importance_sampling_ratio/min": 0.246351420879364, "sampling/importance_sampling_ratio/mean": 1.0542460680007935, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.490487590432167, "clip_ratio/low_mean": 0.12140195537358522, "clip_ratio/low_min": 0.12140195537358522, "clip_ratio/high_mean": 0.03701298777014017, "clip_ratio/high_max": 0.03701298777014017, "clip_ratio/region_mean": 0.1584149431437254, "reward_total_mean": 0.8355494737625122, "reward_meter_mean": 0.9702678918838501, "reward_meter_std": 0.07130211591720581, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.989289402961731, "reward_repeat_soft_std": 0.013245010748505592, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.24833375215530396, "reward_total_composite_mean": 0.8355494737625122, "reward_total_composite_std": 0.1138211116194725} {"timestamp_utc": "2026-04-12T22:54:19Z", "mode": "train", "global_step": 280, "epoch": 0.02812656956303365, "loss": -0.0422, "grad_norm": 3.0571014881134033, "learning_rate": 9.154545454545455e-06, "num_tokens": 542128.0, "completions/mean_length": 232.125, "completions/min_length": 86.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 138.83334350585938, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 269.0, "rewards/meter/mean": 0.5180333256721497, "rewards/meter/std": 0.41655346751213074, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.18898223340511322, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9984127879142761, "rewards/repeat_soft/std": 0.0029342665802687407, "rewards/judge_quality/mean": 0.19875000417232513, "rewards/judge_quality/std": 0.16444168984889984, "rewards/total_composite/mean": 0.41079795360565186, "rewards/total_composite/std": 0.37302470207214355, "reward": 0.41079795360565186, "reward_std": 0.37302467226982117, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20571835339069366, "sampling/sampling_logp_difference/max": 1.9149236679077148, "sampling/importance_sampling_ratio/min": 0.14735308289527893, "sampling/importance_sampling_ratio/mean": 1.0533301830291748, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.883622407913208, "clip_ratio/low_mean": 0.026617100462317467, "clip_ratio/low_min": 0.026617100462317467, "clip_ratio/high_mean": 0.09396354295313358, "clip_ratio/high_max": 0.09396354295313358, "clip_ratio/region_mean": 0.12058064341545105, "reward_total_mean": 0.41079795360565186, "reward_meter_mean": 0.5180333256721497, "reward_meter_std": 0.41655346751213074, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.18898223340511322, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9984127879142761, "reward_repeat_soft_std": 0.0029342665802687407, "reward_judge_quality_mean": 0.19875000417232513, "reward_judge_quality_std": 0.16444168984889984, "reward_total_composite_mean": 0.41079795360565186, "reward_total_composite_std": 0.37302470207214355} {"timestamp_utc": "2026-04-12T22:54:26Z", "mode": "train", "global_step": 281, "epoch": 0.028227021597187343, "loss": 0.1024, "grad_norm": 12.849178314208984, "learning_rate": 9.151515151515153e-06, "num_tokens": 543764.0, "completions/mean_length": 50.5, "completions/min_length": 30.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.5, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.6521459817886353, "rewards/meter/std": 0.421623170375824, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9970614910125732, "rewards/repeat_soft/std": 0.005408369470387697, "rewards/judge_quality/mean": 0.5387499928474426, "rewards/judge_quality/std": 0.24485784769058228, "rewards/total_composite/mean": 0.7047968506813049, "rewards/total_composite/std": 0.20102500915527344, "reward": 0.7047968506813049, "reward_std": 0.20102500915527344, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15823796391487122, "sampling/sampling_logp_difference/max": 1.5390679836273193, "sampling/importance_sampling_ratio/min": 0.21458101272583008, "sampling/importance_sampling_ratio/mean": 1.0135899782180786, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6736142486333847, "clip_ratio/low_mean": 0.035016150679439306, "clip_ratio/low_min": 0.035016150679439306, "clip_ratio/high_mean": 0.09429129958152771, "clip_ratio/high_max": 0.09429129958152771, "clip_ratio/region_mean": 0.12930745026096702, "reward_total_mean": 0.7047968506813049, "reward_meter_mean": 0.6521459817886353, "reward_meter_std": 0.421623170375824, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9970614910125732, "reward_repeat_soft_std": 0.005408369470387697, "reward_judge_quality_mean": 0.5387499928474426, "reward_judge_quality_std": 0.24485784769058228, "reward_total_composite_mean": 0.7047968506813049, "reward_total_composite_std": 0.20102500915527344} {"timestamp_utc": "2026-04-12T22:54:33Z", "mode": "train", "global_step": 282, "epoch": 0.028327473631341034, "loss": -0.0285, "grad_norm": 11.690851211547852, "learning_rate": 9.148484848484848e-06, "num_tokens": 545465.0, "completions/mean_length": 62.625, "completions/min_length": 48.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.625, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.8394278883934021, "rewards/meter/std": 0.25878262519836426, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.993722140789032, "rewards/repeat_soft/std": 0.012258262373507023, "rewards/judge_quality/mean": 0.38874998688697815, "rewards/judge_quality/std": 0.08675704896450043, "rewards/total_composite/mean": 0.7437397241592407, "rewards/total_composite/std": 0.10741536319255829, "reward": 0.7437397241592407, "reward_std": 0.10741537064313889, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19096021354198456, "sampling/sampling_logp_difference/max": 1.5605363845825195, "sampling/importance_sampling_ratio/min": 0.21002337336540222, "sampling/importance_sampling_ratio/mean": 1.0098549127578735, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2616069465875626, "clip_ratio/low_mean": 0.03578628972172737, "clip_ratio/low_min": 0.03578628972172737, "clip_ratio/high_mean": 0.14375481382012367, "clip_ratio/high_max": 0.14375481382012367, "clip_ratio/region_mean": 0.17954110354185104, "reward_total_mean": 0.7437397241592407, "reward_meter_mean": 0.8394278883934021, "reward_meter_std": 0.25878262519836426, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.993722140789032, "reward_repeat_soft_std": 0.012258262373507023, "reward_judge_quality_mean": 0.38874998688697815, "reward_judge_quality_std": 0.08675704896450043, "reward_total_composite_mean": 0.7437397241592407, "reward_total_composite_std": 0.10741536319255829} {"timestamp_utc": "2026-04-12T22:54:44Z", "mode": "train", "global_step": 283, "epoch": 0.028427925665494725, "loss": -0.1928, "grad_norm": 2.7340128421783447, "learning_rate": 9.145454545454546e-06, "num_tokens": 547671.0, "completions/mean_length": 223.75, "completions/min_length": 98.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 127.66667175292969, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 181.0, "rewards/meter/mean": 0.2852289080619812, "rewards/meter/std": 0.24524392187595367, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.14773420989513397, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9934878349304199, "rewards/repeat_soft/std": 0.0055814869701862335, "rewards/judge_quality/mean": 0.26375001668930054, "rewards/judge_quality/std": 0.15361709892749786, "rewards/total_composite/mean": 0.4109269678592682, "rewards/total_composite/std": 0.20993700623512268, "reward": 0.4109269678592682, "reward_std": 0.20993700623512268, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2226027101278305, "sampling/sampling_logp_difference/max": 1.3721113204956055, "sampling/importance_sampling_ratio/min": 0.2535710334777832, "sampling/importance_sampling_ratio/mean": 1.050154209136963, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6940521001815796, "clip_ratio/low_mean": 0.017326733097434044, "clip_ratio/low_min": 0.017326733097434044, "clip_ratio/high_mean": 0.12709505297243595, "clip_ratio/high_max": 0.12709505297243595, "clip_ratio/region_mean": 0.14442178606987, "reward_total_mean": 0.4109269678592682, "reward_meter_mean": 0.2852289080619812, "reward_meter_std": 0.24524392187595367, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.14773420989513397, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9934878349304199, "reward_repeat_soft_std": 0.0055814869701862335, "reward_judge_quality_mean": 0.26375001668930054, "reward_judge_quality_std": 0.15361709892749786, "reward_total_composite_mean": 0.4109269678592682, "reward_total_composite_std": 0.20993700623512268} {"timestamp_utc": "2026-04-12T22:54:52Z", "mode": "train", "global_step": 284, "epoch": 0.028528377699648416, "loss": -0.0045, "grad_norm": 8.279809951782227, "learning_rate": 9.142424242424243e-06, "num_tokens": 549896.0, "completions/mean_length": 103.125, "completions/min_length": 72.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.125, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.19484835863113403, "rewards/meter/std": 0.1526610255241394, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9985024929046631, "rewards/repeat_soft/std": 0.0019619271624833345, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.3765559792518616, "rewards/total_composite/std": 0.17657911777496338, "reward": 0.3765559792518616, "reward_std": 0.17657911777496338, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20476503670215607, "sampling/sampling_logp_difference/max": 1.584421157836914, "sampling/importance_sampling_ratio/min": 0.20506645739078522, "sampling/importance_sampling_ratio/mean": 1.061143159866333, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.340910851955414, "clip_ratio/low_mean": 0.0714705865830183, "clip_ratio/low_min": 0.0714705865830183, "clip_ratio/high_mean": 0.10655750054866076, "clip_ratio/high_max": 0.10655750054866076, "clip_ratio/region_mean": 0.17802808713167906, "reward_total_mean": 0.3765559792518616, "reward_meter_mean": 0.19484835863113403, "reward_meter_std": 0.1526610255241394, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9985024929046631, "reward_repeat_soft_std": 0.0019619271624833345, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.3765559792518616, "reward_total_composite_std": 0.17657911777496338} {"timestamp_utc": "2026-04-12T22:55:03Z", "mode": "train", "global_step": 285, "epoch": 0.02862882973380211, "loss": -0.083, "grad_norm": 3.051809310913086, "learning_rate": 9.13939393939394e-06, "num_tokens": 551388.0, "completions/mean_length": 88.5, "completions/min_length": 19.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 28.000001907348633, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.6935352087020874, "rewards/meter/std": 0.41912996768951416, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9671874642372131, "rewards/repeat_soft/std": 0.013258260674774647, "rewards/judge_quality/mean": 0.3712499737739563, "rewards/judge_quality/std": 0.1470119059085846, "rewards/total_composite/mean": 0.6370596289634705, "rewards/total_composite/std": 0.2911284267902374, "reward": 0.6370596289634705, "reward_std": 0.2911284267902374, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1822977066040039, "sampling/sampling_logp_difference/max": 1.2914104461669922, "sampling/importance_sampling_ratio/min": 0.27488279342651367, "sampling/importance_sampling_ratio/mean": 1.0152482986450195, "sampling/importance_sampling_ratio/max": 1.898888349533081, "entropy": 2.1551780998706818, "clip_ratio/low_mean": 0.02500000037252903, "clip_ratio/low_min": 0.02500000037252903, "clip_ratio/high_mean": 0.16795931570231915, "clip_ratio/high_max": 0.16795931570231915, "clip_ratio/region_mean": 0.19295931607484818, "reward_total_mean": 0.6370596289634705, "reward_meter_mean": 0.6935352087020874, "reward_meter_std": 0.41912996768951416, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9671874642372131, "reward_repeat_soft_std": 0.013258260674774647, "reward_judge_quality_mean": 0.3712499737739563, "reward_judge_quality_std": 0.1470119059085846, "reward_total_composite_mean": 0.6370596289634705, "reward_total_composite_std": 0.2911284267902374} {"timestamp_utc": "2026-04-12T22:55:14Z", "mode": "train", "global_step": 286, "epoch": 0.028729281767955802, "loss": -0.214, "grad_norm": 2.5711779594421387, "learning_rate": 9.136363636363637e-06, "num_tokens": 553633.0, "completions/mean_length": 216.625, "completions/min_length": 73.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 118.16667175292969, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.4555071294307709, "rewards/meter/std": 0.3743753731250763, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9984261989593506, "rewards/repeat_soft/std": 0.0018009545747190714, "rewards/judge_quality/mean": 0.20000000298023224, "rewards/judge_quality/std": 0.09258200973272324, "rewards/total_composite/mean": 0.40172824263572693, "rewards/total_composite/std": 0.2868506610393524, "reward": 0.40172824263572693, "reward_std": 0.2868506610393524, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2060750275850296, "sampling/sampling_logp_difference/max": 1.78460693359375, "sampling/importance_sampling_ratio/min": 0.16786302626132965, "sampling/importance_sampling_ratio/mean": 1.049888014793396, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.55831316113472, "clip_ratio/low_mean": 0.03282815497368574, "clip_ratio/low_min": 0.03282815497368574, "clip_ratio/high_mean": 0.0950759295374155, "clip_ratio/high_max": 0.0950759295374155, "clip_ratio/region_mean": 0.12790408451110125, "reward_total_mean": 0.40172824263572693, "reward_meter_mean": 0.4555071294307709, "reward_meter_std": 0.3743753731250763, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9984261989593506, "reward_repeat_soft_std": 0.0018009545747190714, "reward_judge_quality_mean": 0.20000000298023224, "reward_judge_quality_std": 0.09258200973272324, "reward_total_composite_mean": 0.40172824263572693, "reward_total_composite_std": 0.2868506610393524} {"timestamp_utc": "2026-04-12T22:55:26Z", "mode": "train", "global_step": 287, "epoch": 0.028829733802109493, "loss": -0.1366, "grad_norm": 1.7293504476547241, "learning_rate": 9.133333333333335e-06, "num_tokens": 555489.0, "completions/mean_length": 378.0, "completions/min_length": 117.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.625, "completions/mean_terminated_length": 154.6666717529297, "completions/min_terminated_length": 117.0, "completions/max_terminated_length": 174.0, "rewards/meter/mean": 0.7288732528686523, "rewards/meter/std": 0.36525392532348633, "rewards/count_adherence/mean": 0.7000000476837158, "rewards/count_adherence/std": 0.33806172013282776, "rewards/hard_gate/mean": 0.5, "rewards/hard_gate/std": 0.5345224738121033, "rewards/repeat_soft/mean": 0.9972049593925476, "rewards/repeat_soft/std": 0.002809912199154496, "rewards/judge_quality/mean": 0.1237500011920929, "rewards/judge_quality/std": 0.14181853830814362, "rewards/total_composite/mean": 0.31715723872184753, "rewards/total_composite/std": 0.3544944226741791, "reward": 0.31715723872184753, "reward_std": 0.3544943928718567, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20233936607837677, "sampling/sampling_logp_difference/max": 1.24430513381958, "sampling/importance_sampling_ratio/min": 0.3909626603126526, "sampling/importance_sampling_ratio/mean": 1.0701149702072144, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3634082078933716, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.05959172546863556, "clip_ratio/high_max": 0.05959172546863556, "clip_ratio/region_mean": 0.05959172546863556, "reward_total_mean": 0.31715723872184753, "reward_meter_mean": 0.7288732528686523, "reward_meter_std": 0.36525392532348633, "reward_count_adherence_mean": 0.7000000476837158, "reward_count_adherence_std": 0.33806172013282776, "reward_hard_gate_mean": 0.5, "reward_hard_gate_std": 0.5345224738121033, "reward_repeat_soft_mean": 0.9972049593925476, "reward_repeat_soft_std": 0.002809912199154496, "reward_judge_quality_mean": 0.1237500011920929, "reward_judge_quality_std": 0.14181853830814362, "reward_total_composite_mean": 0.31715723872184753, "reward_total_composite_std": 0.3544944226741791} {"timestamp_utc": "2026-04-12T22:55:32Z", "mode": "train", "global_step": 288, "epoch": 0.028930185836263184, "loss": 0.0501, "grad_norm": 14.6194486618042, "learning_rate": 9.130303030303032e-06, "num_tokens": 557114.0, "completions/mean_length": 48.125, "completions/min_length": 36.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.125, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9153883457183838, "rewards/meter/std": 0.13079427182674408, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9871237277984619, "rewards/repeat_soft/std": 0.02694631554186344, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.0843462198972702, "rewards/total_composite/mean": 0.776137113571167, "rewards/total_composite/std": 0.06920462846755981, "reward": 0.776137113571167, "reward_std": 0.06920463591814041, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1859467476606369, "sampling/sampling_logp_difference/max": 1.1919422149658203, "sampling/importance_sampling_ratio/min": 0.3036309778690338, "sampling/importance_sampling_ratio/mean": 1.051623821258545, "sampling/importance_sampling_ratio/max": 1.829160213470459, "entropy": 2.5691657066345215, "clip_ratio/low_mean": 0.048171233385801315, "clip_ratio/low_min": 0.048171233385801315, "clip_ratio/high_mean": 0.11287985742092133, "clip_ratio/high_max": 0.11287985742092133, "clip_ratio/region_mean": 0.16105109080672264, "reward_total_mean": 0.776137113571167, "reward_meter_mean": 0.9153883457183838, "reward_meter_std": 0.13079427182674408, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9871237277984619, "reward_repeat_soft_std": 0.02694631554186344, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.0843462198972702, "reward_total_composite_mean": 0.776137113571167, "reward_total_composite_std": 0.06920462846755981} {"timestamp_utc": "2026-04-12T22:55:39Z", "mode": "train", "global_step": 289, "epoch": 0.029030637870416875, "loss": -0.016, "grad_norm": 10.922988891601562, "learning_rate": 9.127272727272727e-06, "num_tokens": 559050.0, "completions/mean_length": 78.0, "completions/min_length": 62.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.4784967601299286, "rewards/meter/std": 0.2557385265827179, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9909804463386536, "rewards/repeat_soft/std": 0.00834750197827816, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.08166787773370743, "rewards/total_composite/mean": 0.5741090774536133, "rewards/total_composite/std": 0.1284409910440445, "reward": 0.5741090774536133, "reward_std": 0.1284409910440445, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1858014315366745, "sampling/sampling_logp_difference/max": 2.033993721008301, "sampling/importance_sampling_ratio/min": 0.13081204891204834, "sampling/importance_sampling_ratio/mean": 1.0157337188720703, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4662089496850967, "clip_ratio/low_mean": 0.09549055993556976, "clip_ratio/low_min": 0.09549055993556976, "clip_ratio/high_mean": 0.09804266691207886, "clip_ratio/high_max": 0.09804266691207886, "clip_ratio/region_mean": 0.19353322684764862, "reward_total_mean": 0.5741090774536133, "reward_meter_mean": 0.4784967601299286, "reward_meter_std": 0.2557385265827179, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9909804463386536, "reward_repeat_soft_std": 0.00834750197827816, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.08166787773370743, "reward_total_composite_mean": 0.5741090774536133, "reward_total_composite_std": 0.1284409910440445} {"timestamp_utc": "2026-04-12T22:55:45Z", "mode": "train", "global_step": 290, "epoch": 0.029131089904570567, "loss": 0.0605, "grad_norm": 11.17509651184082, "learning_rate": 9.124242424242425e-06, "num_tokens": 560852.0, "completions/mean_length": 59.25, "completions/min_length": 48.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.25, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9032805562019348, "rewards/meter/std": 0.14097441732883453, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9983168244361877, "rewards/repeat_soft/std": 0.003002339508384466, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.8021829128265381, "rewards/total_composite/std": 0.09201550483703613, "reward": 0.8021829128265381, "reward_std": 0.09201550483703613, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1700136512517929, "sampling/sampling_logp_difference/max": 1.333479881286621, "sampling/importance_sampling_ratio/min": 0.2635585069656372, "sampling/importance_sampling_ratio/mean": 1.0300601720809937, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1631518602371216, "clip_ratio/low_mean": 0.04913408402353525, "clip_ratio/low_min": 0.04913408402353525, "clip_ratio/high_mean": 0.10975005757063627, "clip_ratio/high_max": 0.10975005757063627, "clip_ratio/region_mean": 0.15888414159417152, "reward_total_mean": 0.8021829128265381, "reward_meter_mean": 0.9032805562019348, "reward_meter_std": 0.14097441732883453, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9983168244361877, "reward_repeat_soft_std": 0.003002339508384466, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.8021829128265381, "reward_total_composite_std": 0.09201550483703613} {"timestamp_utc": "2026-04-12T22:55:52Z", "mode": "train", "global_step": 291, "epoch": 0.029231541938724258, "loss": -0.0761, "grad_norm": 13.153314590454102, "learning_rate": 9.121212121212122e-06, "num_tokens": 562717.0, "completions/mean_length": 56.125, "completions/min_length": 41.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.125, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.7781696319580078, "rewards/meter/std": 0.38469940423965454, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9976159930229187, "rewards/repeat_soft/std": 0.005910256877541542, "rewards/judge_quality/mean": 0.32999998331069946, "rewards/judge_quality/std": 0.11747339367866516, "rewards/total_composite/mean": 0.6989379525184631, "rewards/total_composite/std": 0.17578941583633423, "reward": 0.6989379525184631, "reward_std": 0.17578940093517303, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18136155605316162, "sampling/sampling_logp_difference/max": 1.6687707901000977, "sampling/importance_sampling_ratio/min": 0.18847858905792236, "sampling/importance_sampling_ratio/mean": 1.0269290208816528, "sampling/importance_sampling_ratio/max": 1.9873640537261963, "entropy": 2.1261725276708603, "clip_ratio/low_mean": 0.02385222353041172, "clip_ratio/low_min": 0.02385222353041172, "clip_ratio/high_mean": 0.11965310666710138, "clip_ratio/high_max": 0.11965310666710138, "clip_ratio/region_mean": 0.1435053301975131, "reward_total_mean": 0.6989379525184631, "reward_meter_mean": 0.7781696319580078, "reward_meter_std": 0.38469940423965454, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9976159930229187, "reward_repeat_soft_std": 0.005910256877541542, "reward_judge_quality_mean": 0.32999998331069946, "reward_judge_quality_std": 0.11747339367866516, "reward_total_composite_mean": 0.6989379525184631, "reward_total_composite_std": 0.17578941583633423} {"timestamp_utc": "2026-04-12T22:55:59Z", "mode": "train", "global_step": 292, "epoch": 0.029331993972877952, "loss": 0.0744, "grad_norm": 8.58546257019043, "learning_rate": 9.118181818181819e-06, "num_tokens": 565250.0, "completions/mean_length": 135.625, "completions/min_length": 95.0, "completions/max_length": 151.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 135.625, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 151.0, "rewards/meter/mean": 0.5545876026153564, "rewards/meter/std": 0.2960638105869293, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.987626314163208, "rewards/repeat_soft/std": 0.006403167266398668, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.6179520487785339, "rewards/total_composite/std": 0.14281821250915527, "reward": 0.6179520487785339, "reward_std": 0.14281822741031647, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17466145753860474, "sampling/sampling_logp_difference/max": 1.131396770477295, "sampling/importance_sampling_ratio/min": 0.3225823640823364, "sampling/importance_sampling_ratio/mean": 1.0307008028030396, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9526078402996063, "clip_ratio/low_mean": 0.0831034379079938, "clip_ratio/low_min": 0.0831034379079938, "clip_ratio/high_mean": 0.08646366000175476, "clip_ratio/high_max": 0.08646366000175476, "clip_ratio/region_mean": 0.16956709790974855, "reward_total_mean": 0.6179520487785339, "reward_meter_mean": 0.5545876026153564, "reward_meter_std": 0.2960638105869293, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.987626314163208, "reward_repeat_soft_std": 0.006403167266398668, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.6179520487785339, "reward_total_composite_std": 0.14281821250915527} {"timestamp_utc": "2026-04-12T22:56:11Z", "mode": "train", "global_step": 293, "epoch": 0.029432446007031644, "loss": -0.1025, "grad_norm": 3.7004501819610596, "learning_rate": 9.115151515151516e-06, "num_tokens": 566937.0, "completions/mean_length": 99.875, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 41.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.471249520778656, "rewards/meter/std": 0.3917904496192932, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9947205185890198, "rewards/repeat_soft/std": 0.010595371015369892, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.5405343174934387, "rewards/total_composite/std": 0.2676551640033722, "reward": 0.5405343174934387, "reward_std": 0.2676551640033722, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20599544048309326, "sampling/sampling_logp_difference/max": 1.1931915283203125, "sampling/importance_sampling_ratio/min": 0.30325186252593994, "sampling/importance_sampling_ratio/mean": 1.0507829189300537, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.387362837791443, "clip_ratio/low_mean": 0.05929487384855747, "clip_ratio/low_min": 0.05929487384855747, "clip_ratio/high_mean": 0.10129657480865717, "clip_ratio/high_max": 0.10129657480865717, "clip_ratio/region_mean": 0.16059144865721464, "reward_total_mean": 0.5405343174934387, "reward_meter_mean": 0.471249520778656, "reward_meter_std": 0.3917904496192932, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9947205185890198, "reward_repeat_soft_std": 0.010595371015369892, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.5405343174934387, "reward_total_composite_std": 0.2676551640033722} {"timestamp_utc": "2026-04-12T22:56:18Z", "mode": "train", "global_step": 294, "epoch": 0.029532898041185335, "loss": 0.0496, "grad_norm": 10.780856132507324, "learning_rate": 9.112121212121214e-06, "num_tokens": 568967.0, "completions/mean_length": 75.75, "completions/min_length": 37.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.75, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.7983524799346924, "rewards/meter/std": 0.2799713611602783, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9893368482589722, "rewards/repeat_soft/std": 0.009097153320908546, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.2424134910106659, "rewards/total_composite/mean": 0.7601922750473022, "rewards/total_composite/std": 0.18512994050979614, "reward": 0.7601922750473022, "reward_std": 0.18512992560863495, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17501911520957947, "sampling/sampling_logp_difference/max": 1.294020652770996, "sampling/importance_sampling_ratio/min": 0.2741662263870239, "sampling/importance_sampling_ratio/mean": 1.0327290296554565, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0156541764736176, "clip_ratio/low_mean": 0.0289523396641016, "clip_ratio/low_min": 0.0289523396641016, "clip_ratio/high_mean": 0.12474126648157835, "clip_ratio/high_max": 0.12474126648157835, "clip_ratio/region_mean": 0.15369360614567995, "reward_total_mean": 0.7601922750473022, "reward_meter_mean": 0.7983524799346924, "reward_meter_std": 0.2799713611602783, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9893368482589722, "reward_repeat_soft_std": 0.009097153320908546, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.2424134910106659, "reward_total_composite_mean": 0.7601922750473022, "reward_total_composite_std": 0.18512994050979614} {"timestamp_utc": "2026-04-12T22:56:24Z", "mode": "train", "global_step": 295, "epoch": 0.029633350075339026, "loss": 0.0311, "grad_norm": 17.212636947631836, "learning_rate": 9.10909090909091e-06, "num_tokens": 570431.0, "completions/mean_length": 29.0, "completions/min_length": 16.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.772325873374939, "rewards/meter/std": 0.3879622220993042, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9593343734741211, "rewards/repeat_soft/std": 0.008953613229095936, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.7967300415039062, "rewards/total_composite/std": 0.22701998054981232, "reward": 0.7967300415039062, "reward_std": 0.22701998054981232, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12177182734012604, "sampling/sampling_logp_difference/max": 1.378462553024292, "sampling/importance_sampling_ratio/min": 0.25196564197540283, "sampling/importance_sampling_ratio/mean": 1.0237075090408325, "sampling/importance_sampling_ratio/max": 1.875808596611023, "entropy": 0.8628395423293114, "clip_ratio/low_mean": 0.019318182487040758, "clip_ratio/low_min": 0.019318182487040758, "clip_ratio/high_mean": 0.09079835750162601, "clip_ratio/high_max": 0.09079835750162601, "clip_ratio/region_mean": 0.11011653998866677, "reward_total_mean": 0.7967300415039062, "reward_meter_mean": 0.772325873374939, "reward_meter_std": 0.3879622220993042, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9593343734741211, "reward_repeat_soft_std": 0.008953613229095936, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.7967300415039062, "reward_total_composite_std": 0.22701998054981232} {"timestamp_utc": "2026-04-12T22:56:30Z", "mode": "train", "global_step": 296, "epoch": 0.029733802109492717, "loss": 0.0472, "grad_norm": 14.761397361755371, "learning_rate": 9.106060606060606e-06, "num_tokens": 571795.0, "completions/mean_length": 30.5, "completions/min_length": 22.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.5, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.6134821176528931, "rewards/meter/std": 0.44915860891342163, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9546234011650085, "rewards/repeat_soft/std": 0.022278277203440666, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.2332380712032318, "rewards/total_composite/mean": 0.6445292830467224, "rewards/total_composite/std": 0.22291666269302368, "reward": 0.6445292830467224, "reward_std": 0.22291667759418488, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19703973829746246, "sampling/sampling_logp_difference/max": 1.1205410957336426, "sampling/importance_sampling_ratio/min": 0.3261032998561859, "sampling/importance_sampling_ratio/mean": 1.060241937637329, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9870200604200363, "clip_ratio/low_mean": 0.05626114271581173, "clip_ratio/low_min": 0.05626114271581173, "clip_ratio/high_mean": 0.051748512079939246, "clip_ratio/high_max": 0.051748512079939246, "clip_ratio/region_mean": 0.10800965479575098, "reward_total_mean": 0.6445292830467224, "reward_meter_mean": 0.6134821176528931, "reward_meter_std": 0.44915860891342163, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9546234011650085, "reward_repeat_soft_std": 0.022278277203440666, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.2332380712032318, "reward_total_composite_mean": 0.6445292830467224, "reward_total_composite_std": 0.22291666269302368} {"timestamp_utc": "2026-04-12T22:56:37Z", "mode": "train", "global_step": 297, "epoch": 0.02983425414364641, "loss": -0.0236, "grad_norm": 8.24355411529541, "learning_rate": 9.103030303030304e-06, "num_tokens": 573777.0, "completions/mean_length": 92.75, "completions/min_length": 67.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.75, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.7066262364387512, "rewards/meter/std": 0.42695197463035583, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9860045909881592, "rewards/repeat_soft/std": 0.015309811569750309, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.6974572539329529, "rewards/total_composite/std": 0.19878838956356049, "reward": 0.6974572539329529, "reward_std": 0.19878840446472168, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16979046165943146, "sampling/sampling_logp_difference/max": 1.0780458450317383, "sampling/importance_sampling_ratio/min": 0.3402597904205322, "sampling/importance_sampling_ratio/mean": 1.0493923425674438, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.545069709420204, "clip_ratio/low_mean": 0.05440819729119539, "clip_ratio/low_min": 0.05440819729119539, "clip_ratio/high_mean": 0.09488957189023495, "clip_ratio/high_max": 0.09488957189023495, "clip_ratio/region_mean": 0.14929776918143034, "reward_total_mean": 0.6974572539329529, "reward_meter_mean": 0.7066262364387512, "reward_meter_std": 0.42695197463035583, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9860045909881592, "reward_repeat_soft_std": 0.015309811569750309, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.6974572539329529, "reward_total_composite_std": 0.19878838956356049} {"timestamp_utc": "2026-04-12T22:56:46Z", "mode": "train", "global_step": 298, "epoch": 0.0299347061778001, "loss": 0.0852, "grad_norm": 13.311131477355957, "learning_rate": 9.100000000000001e-06, "num_tokens": 575911.0, "completions/mean_length": 77.75, "completions/min_length": 47.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.75, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.6634586453437805, "rewards/meter/std": 0.31895914673805237, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9711800813674927, "rewards/repeat_soft/std": 0.01637611724436283, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.6765493750572205, "rewards/total_composite/std": 0.17022906243801117, "reward": 0.6765493750572205, "reward_std": 0.17022904753684998, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1438024938106537, "sampling/sampling_logp_difference/max": 1.4693100452423096, "sampling/importance_sampling_ratio/min": 0.23008416593074799, "sampling/importance_sampling_ratio/mean": 1.0232383012771606, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0774861872196198, "clip_ratio/low_mean": 0.07356991898268461, "clip_ratio/low_min": 0.07356991898268461, "clip_ratio/high_mean": 0.08596846554428339, "clip_ratio/high_max": 0.08596846554428339, "clip_ratio/region_mean": 0.159538384526968, "reward_total_mean": 0.6765493750572205, "reward_meter_mean": 0.6634586453437805, "reward_meter_std": 0.31895914673805237, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9711800813674927, "reward_repeat_soft_std": 0.01637611724436283, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.6765493750572205, "reward_total_composite_std": 0.17022906243801117} {"timestamp_utc": "2026-04-12T22:56:53Z", "mode": "train", "global_step": 299, "epoch": 0.03003515821195379, "loss": 0.0527, "grad_norm": 7.654555797576904, "learning_rate": 9.096969696969698e-06, "num_tokens": 578027.0, "completions/mean_length": 92.5, "completions/min_length": 61.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.5, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.7970134019851685, "rewards/meter/std": 0.36591342091560364, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9632469415664673, "rewards/repeat_soft/std": 0.04446980729699135, "rewards/judge_quality/mean": 0.3974999785423279, "rewards/judge_quality/std": 0.15745748579502106, "rewards/total_composite/mean": 0.7242307662963867, "rewards/total_composite/std": 0.18226881325244904, "reward": 0.7242307662963867, "reward_std": 0.18226881325244904, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17630302906036377, "sampling/sampling_logp_difference/max": 1.2100534439086914, "sampling/importance_sampling_ratio/min": 0.29818135499954224, "sampling/importance_sampling_ratio/mean": 1.055213451385498, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.149080604314804, "clip_ratio/low_mean": 0.021465029567480087, "clip_ratio/low_min": 0.021465029567480087, "clip_ratio/high_mean": 0.11708532366901636, "clip_ratio/high_max": 0.11708532366901636, "clip_ratio/region_mean": 0.13855035323649645, "reward_total_mean": 0.7242307662963867, "reward_meter_mean": 0.7970134019851685, "reward_meter_std": 0.36591342091560364, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9632469415664673, "reward_repeat_soft_std": 0.04446980729699135, "reward_judge_quality_mean": 0.3974999785423279, "reward_judge_quality_std": 0.15745748579502106, "reward_total_composite_mean": 0.7242307662963867, "reward_total_composite_std": 0.18226881325244904} {"timestamp_utc": "2026-04-12T22:56:59Z", "mode": "train", "global_step": 300, "epoch": 0.030135610246107485, "loss": 0.0128, "grad_norm": 10.625736236572266, "learning_rate": 9.093939393939395e-06, "num_tokens": 579731.0, "completions/mean_length": 58.0, "completions/min_length": 52.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.8298797607421875, "rewards/meter/std": 0.335178017616272, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9931271076202393, "rewards/repeat_soft/std": 0.012673873454332352, "rewards/judge_quality/mean": 0.3999999761581421, "rewards/judge_quality/std": 0.09258200973272324, "rewards/total_composite/mean": 0.6651860475540161, "rewards/total_composite/std": 0.3057008385658264, "reward": 0.6651860475540161, "reward_std": 0.3057008385658264, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1634523719549179, "sampling/sampling_logp_difference/max": 1.458099365234375, "sampling/importance_sampling_ratio/min": 0.23267808556556702, "sampling/importance_sampling_ratio/mean": 1.0294554233551025, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4084550440311432, "clip_ratio/low_mean": 0.035566188395023346, "clip_ratio/low_min": 0.035566188395023346, "clip_ratio/high_mean": 0.12594011332839727, "clip_ratio/high_max": 0.12594011332839727, "clip_ratio/region_mean": 0.16150630172342062, "reward_total_mean": 0.6651860475540161, "reward_meter_mean": 0.8298797607421875, "reward_meter_std": 0.335178017616272, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9931271076202393, "reward_repeat_soft_std": 0.012673873454332352, "reward_judge_quality_mean": 0.3999999761581421, "reward_judge_quality_std": 0.09258200973272324, "reward_total_composite_mean": 0.6651860475540161, "reward_total_composite_std": 0.3057008385658264} {"timestamp_utc": "2026-04-12T22:57:54Z", "mode": "eval", "global_step": 300, "epoch": 0.030135610246107485, "eval_loss": NaN, "eval_runtime": 54.6657, "eval_samples_per_second": 1.463, "eval_steps_per_second": 0.183, "eval_num_tokens": 579731.0, "eval_completions/mean_length": 101.45, "eval_completions/min_length": 34.9, "eval_completions/max_length": 229.0, "eval_completions/clipped_ratio": 0.0375, "eval_completions/mean_terminated_length": 85.05833358764649, "eval_completions/min_terminated_length": 34.9, "eval_completions/max_terminated_length": 152.8, "eval_rewards/meter/mean": 0.6883099973201752, "eval_rewards/meter/std": 0.3315569184720516, "eval_rewards/count_adherence/mean": 0.9804166615009308, "eval_rewards/count_adherence/std": 0.04410490021109581, "eval_rewards/hard_gate/mean": 0.9375, "eval_rewards/hard_gate/std": 0.09804592728614807, "eval_rewards/repeat_soft/mean": 0.9900646924972534, "eval_rewards/repeat_soft/std": 0.01001640916801989, "eval_rewards/judge_quality/mean": 0.439124995470047, "eval_rewards/judge_quality/std": 0.16539452522993087, "eval_rewards/total_composite/mean": 0.6526301890611649, "eval_rewards/total_composite/std": 0.20698279999196528, "eval_reward": 0.6526301890611649, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.12535401359200476, "eval_sampling/sampling_logp_difference/max": 1.0039912223815919, "eval_sampling/importance_sampling_ratio/min": 0.3699460715055466, "eval_sampling/importance_sampling_ratio/mean": 1.034483802318573, "eval_sampling/importance_sampling_ratio/max": 1.566494607925415, "eval_entropy": 1.9639609456062317, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6526301890611649, "eval_reward_meter_mean": 0.6883099973201752, "eval_reward_meter_std": 0.3315569184720516, "eval_reward_count_adherence_mean": 0.9804166615009308, "eval_reward_count_adherence_std": 0.04410490021109581, "eval_reward_hard_gate_mean": 0.9375, "eval_reward_hard_gate_std": 0.09804592728614807, "eval_reward_repeat_soft_mean": 0.9900646924972534, "eval_reward_repeat_soft_std": 0.01001640916801989, "eval_reward_judge_quality_mean": 0.439124995470047, "eval_reward_judge_quality_std": 0.16539452522993087, "eval_reward_total_composite_mean": 0.6526301890611649, "eval_reward_total_composite_std": 0.20698279999196528} {"timestamp_utc": "2026-04-12T22:58:04Z", "mode": "train", "global_step": 301, "epoch": 0.030236062280261176, "loss": 0.1167, "grad_norm": 10.304308891296387, "learning_rate": 9.090909090909091e-06, "num_tokens": 581895.0, "completions/mean_length": 92.5, "completions/min_length": 64.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.5, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.5853999853134155, "rewards/meter/std": 0.37097206711769104, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9947682619094849, "rewards/repeat_soft/std": 0.005658273585140705, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.6261568069458008, "rewards/total_composite/std": 0.1574321985244751, "reward": 0.6261568069458008, "reward_std": 0.15743222832679749, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20102104544639587, "sampling/sampling_logp_difference/max": 2.242929458618164, "sampling/importance_sampling_ratio/min": 0.10614710301160812, "sampling/importance_sampling_ratio/mean": 1.040395736694336, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5712140947580338, "clip_ratio/low_mean": 0.07154787611216307, "clip_ratio/low_min": 0.07154787611216307, "clip_ratio/high_mean": 0.09798869863152504, "clip_ratio/high_max": 0.09798869863152504, "clip_ratio/region_mean": 0.1695365747436881, "reward_total_mean": 0.6261568069458008, "reward_meter_mean": 0.5853999853134155, "reward_meter_std": 0.37097206711769104, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9947682619094849, "reward_repeat_soft_std": 0.005658273585140705, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.6261568069458008, "reward_total_composite_std": 0.1574321985244751} {"timestamp_utc": "2026-04-12T22:58:11Z", "mode": "train", "global_step": 302, "epoch": 0.030336514314414868, "loss": 0.0328, "grad_norm": 7.513288974761963, "learning_rate": 9.087878787878788e-06, "num_tokens": 584014.0, "completions/mean_length": 93.875, "completions/min_length": 80.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.875, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.987905740737915, "rewards/meter/std": 0.023070070892572403, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.986316442489624, "rewards/repeat_soft/std": 0.01963157393038273, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.7899392247200012, "rewards/total_composite/std": 0.0316375233232975, "reward": 0.7899392247200012, "reward_std": 0.0316375233232975, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16265951097011566, "sampling/sampling_logp_difference/max": 1.5849685668945312, "sampling/importance_sampling_ratio/min": 0.20495423674583435, "sampling/importance_sampling_ratio/mean": 1.0330413579940796, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.187642902135849, "clip_ratio/low_mean": 0.05470214132219553, "clip_ratio/low_min": 0.05470214132219553, "clip_ratio/high_mean": 0.06937852967530489, "clip_ratio/high_max": 0.06937852967530489, "clip_ratio/region_mean": 0.12408067099750042, "reward_total_mean": 0.7899392247200012, "reward_meter_mean": 0.987905740737915, "reward_meter_std": 0.023070070892572403, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.986316442489624, "reward_repeat_soft_std": 0.01963157393038273, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.7899392247200012, "reward_total_composite_std": 0.0316375233232975} {"timestamp_utc": "2026-04-12T22:58:18Z", "mode": "train", "global_step": 303, "epoch": 0.03043696634856856, "loss": -0.0404, "grad_norm": 8.049524307250977, "learning_rate": 9.084848484848486e-06, "num_tokens": 586625.0, "completions/mean_length": 140.375, "completions/min_length": 90.0, "completions/max_length": 180.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 140.375, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 180.0, "rewards/meter/mean": 0.9034425616264343, "rewards/meter/std": 0.1863163858652115, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9586039781570435, "rewards/repeat_soft/std": 0.10051602870225906, "rewards/judge_quality/mean": 0.29249998927116394, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7364095449447632, "rewards/total_composite/std": 0.0880313515663147, "reward": 0.7364095449447632, "reward_std": 0.0880313441157341, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1683201789855957, "sampling/sampling_logp_difference/max": 1.6334912776947021, "sampling/importance_sampling_ratio/min": 0.19524671137332916, "sampling/importance_sampling_ratio/mean": 1.0344997644424438, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8998944014310837, "clip_ratio/low_mean": 0.02873343462124467, "clip_ratio/low_min": 0.02873343462124467, "clip_ratio/high_mean": 0.14088916406035423, "clip_ratio/high_max": 0.14088916406035423, "clip_ratio/region_mean": 0.1696225986815989, "reward_total_mean": 0.7364095449447632, "reward_meter_mean": 0.9034425616264343, "reward_meter_std": 0.1863163858652115, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9586039781570435, "reward_repeat_soft_std": 0.10051602870225906, "reward_judge_quality_mean": 0.29249998927116394, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7364095449447632, "reward_total_composite_std": 0.0880313515663147} {"timestamp_utc": "2026-04-12T22:58:28Z", "mode": "train", "global_step": 304, "epoch": 0.03053741838272225, "loss": -0.0275, "grad_norm": 7.077108383178711, "learning_rate": 9.081818181818183e-06, "num_tokens": 589165.0, "completions/mean_length": 139.5, "completions/min_length": 81.0, "completions/max_length": 163.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 139.5, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 163.0, "rewards/meter/mean": 0.5122875571250916, "rewards/meter/std": 0.3204350173473358, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9933366775512695, "rewards/repeat_soft/std": 0.0075309546664357185, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.57307368516922, "rewards/total_composite/std": 0.2837498188018799, "reward": 0.57307368516922, "reward_std": 0.2837498188018799, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15441834926605225, "sampling/sampling_logp_difference/max": 2.987614631652832, "sampling/importance_sampling_ratio/min": 0.0504075326025486, "sampling/importance_sampling_ratio/mean": 1.0107468366622925, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2485113516449928, "clip_ratio/low_mean": 0.05864398367702961, "clip_ratio/low_min": 0.05864398367702961, "clip_ratio/high_mean": 0.09656731970608234, "clip_ratio/high_max": 0.09656731970608234, "clip_ratio/region_mean": 0.15521130338311195, "reward_total_mean": 0.57307368516922, "reward_meter_mean": 0.5122875571250916, "reward_meter_std": 0.3204350173473358, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9933366775512695, "reward_repeat_soft_std": 0.0075309546664357185, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.57307368516922, "reward_total_composite_std": 0.2837498188018799} {"timestamp_utc": "2026-04-12T22:58:34Z", "mode": "train", "global_step": 305, "epoch": 0.03063787041687594, "loss": 0.0804, "grad_norm": 16.56997299194336, "learning_rate": 9.078787878787878e-06, "num_tokens": 590627.0, "completions/mean_length": 37.75, "completions/min_length": 32.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.75, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.506746232509613, "rewards/meter/std": 0.32195544242858887, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9934844970703125, "rewards/repeat_soft/std": 0.010391284711658955, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.6618842482566833, "rewards/total_composite/std": 0.18032899498939514, "reward": 0.6618842482566833, "reward_std": 0.18032899498939514, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20984283089637756, "sampling/sampling_logp_difference/max": 1.229973554611206, "sampling/importance_sampling_ratio/min": 0.30913713574409485, "sampling/importance_sampling_ratio/mean": 1.07157301902771, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3337265253067017, "clip_ratio/low_mean": 0.11533613782376051, "clip_ratio/low_min": 0.11533613782376051, "clip_ratio/high_mean": 0.10550875775516033, "clip_ratio/high_max": 0.10550875775516033, "clip_ratio/region_mean": 0.22084489557892084, "reward_total_mean": 0.6618842482566833, "reward_meter_mean": 0.506746232509613, "reward_meter_std": 0.32195544242858887, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9934844970703125, "reward_repeat_soft_std": 0.010391284711658955, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.6618842482566833, "reward_total_composite_std": 0.18032899498939514} {"timestamp_utc": "2026-04-12T22:58:41Z", "mode": "train", "global_step": 306, "epoch": 0.030738322451029632, "loss": -0.0023, "grad_norm": 11.527295112609863, "learning_rate": 9.075757575757577e-06, "num_tokens": 592748.0, "completions/mean_length": 94.125, "completions/min_length": 66.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.125, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.7555601596832275, "rewards/meter/std": 0.28098395466804504, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9926422238349915, "rewards/repeat_soft/std": 0.01123738568276167, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.7377663254737854, "rewards/total_composite/std": 0.14512281119823456, "reward": 0.7377663254737854, "reward_std": 0.14512281119823456, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1843918263912201, "sampling/sampling_logp_difference/max": 1.941117286682129, "sampling/importance_sampling_ratio/min": 0.14354348182678223, "sampling/importance_sampling_ratio/mean": 1.0559049844741821, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.079297959804535, "clip_ratio/low_mean": 0.04704944230616093, "clip_ratio/low_min": 0.04704944230616093, "clip_ratio/high_mean": 0.12031625676900148, "clip_ratio/high_max": 0.12031625676900148, "clip_ratio/region_mean": 0.1673656990751624, "reward_total_mean": 0.7377663254737854, "reward_meter_mean": 0.7555601596832275, "reward_meter_std": 0.28098395466804504, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9926422238349915, "reward_repeat_soft_std": 0.01123738568276167, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.7377663254737854, "reward_total_composite_std": 0.14512281119823456} {"timestamp_utc": "2026-04-12T22:58:47Z", "mode": "train", "global_step": 307, "epoch": 0.030838774485183323, "loss": 0.0286, "grad_norm": 11.365004539489746, "learning_rate": 9.072727272727273e-06, "num_tokens": 594317.0, "completions/mean_length": 54.125, "completions/min_length": 52.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.125, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.5253753662109375, "rewards/meter/std": 0.39250195026397705, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9954754114151001, "rewards/repeat_soft/std": 0.006678203586488962, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.19255799055099487, "rewards/total_composite/mean": 0.6277164816856384, "rewards/total_composite/std": 0.2010253220796585, "reward": 0.6277164816856384, "reward_std": 0.2010253220796585, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18113966286182404, "sampling/sampling_logp_difference/max": 1.345733642578125, "sampling/importance_sampling_ratio/min": 0.2603486478328705, "sampling/importance_sampling_ratio/mean": 1.0436046123504639, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7921406030654907, "clip_ratio/low_mean": 0.13858510367572308, "clip_ratio/low_min": 0.13858510367572308, "clip_ratio/high_mean": 0.06493199896067381, "clip_ratio/high_max": 0.06493199896067381, "clip_ratio/region_mean": 0.20351710263639688, "reward_total_mean": 0.6277164816856384, "reward_meter_mean": 0.5253753662109375, "reward_meter_std": 0.39250195026397705, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9954754114151001, "reward_repeat_soft_std": 0.006678203586488962, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.19255799055099487, "reward_total_composite_mean": 0.6277164816856384, "reward_total_composite_std": 0.2010253220796585} {"timestamp_utc": "2026-04-12T22:58:54Z", "mode": "train", "global_step": 308, "epoch": 0.030939226519337018, "loss": 0.1347, "grad_norm": 12.413131713867188, "learning_rate": 9.06969696969697e-06, "num_tokens": 596177.0, "completions/mean_length": 47.5, "completions/min_length": 32.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.5, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9766214489936829, "rewards/meter/std": 0.03410787880420685, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9962619543075562, "rewards/repeat_soft/std": 0.004523929208517075, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.1940544992685318, "rewards/total_composite/mean": 0.819230854511261, "rewards/total_composite/std": 0.0756574496626854, "reward": 0.819230854511261, "reward_std": 0.07565745711326599, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1693844050168991, "sampling/sampling_logp_difference/max": 1.451399326324463, "sampling/importance_sampling_ratio/min": 0.2342422604560852, "sampling/importance_sampling_ratio/mean": 1.0358800888061523, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1998902559280396, "clip_ratio/low_mean": 0.08678034693002701, "clip_ratio/low_min": 0.08678034693002701, "clip_ratio/high_mean": 0.08526629395782948, "clip_ratio/high_max": 0.08526629395782948, "clip_ratio/region_mean": 0.17204664088785648, "reward_total_mean": 0.819230854511261, "reward_meter_mean": 0.9766214489936829, "reward_meter_std": 0.03410787880420685, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9962619543075562, "reward_repeat_soft_std": 0.004523929208517075, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.1940544992685318, "reward_total_composite_mean": 0.819230854511261, "reward_total_composite_std": 0.0756574496626854} {"timestamp_utc": "2026-04-12T22:59:06Z", "mode": "train", "global_step": 309, "epoch": 0.03103967855349071, "loss": -0.2102, "grad_norm": 2.2145590782165527, "learning_rate": 9.066666666666667e-06, "num_tokens": 598548.0, "completions/mean_length": 172.375, "completions/min_length": 116.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 123.85714721679688, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.7892868518829346, "rewards/meter/std": 0.2912038564682007, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9920827150344849, "rewards/repeat_soft/std": 0.005913217086344957, "rewards/judge_quality/mean": 0.33124998211860657, "rewards/judge_quality/std": 0.13715866208076477, "rewards/total_composite/mean": 0.6601037979125977, "rewards/total_composite/std": 0.27576544880867004, "reward": 0.6601037979125977, "reward_std": 0.27576547861099243, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1617843061685562, "sampling/sampling_logp_difference/max": 1.7526416778564453, "sampling/importance_sampling_ratio/min": 0.17331549525260925, "sampling/importance_sampling_ratio/mean": 1.0389899015426636, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.587126910686493, "clip_ratio/low_mean": 0.017045455053448677, "clip_ratio/low_min": 0.017045455053448677, "clip_ratio/high_mean": 0.10600725747644901, "clip_ratio/high_max": 0.10600725747644901, "clip_ratio/region_mean": 0.12305271252989769, "reward_total_mean": 0.6601037979125977, "reward_meter_mean": 0.7892868518829346, "reward_meter_std": 0.2912038564682007, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9920827150344849, "reward_repeat_soft_std": 0.005913217086344957, "reward_judge_quality_mean": 0.33124998211860657, "reward_judge_quality_std": 0.13715866208076477, "reward_total_composite_mean": 0.6601037979125977, "reward_total_composite_std": 0.27576544880867004} {"timestamp_utc": "2026-04-12T22:59:15Z", "mode": "train", "global_step": 310, "epoch": 0.0311401305876444, "loss": 0.0489, "grad_norm": 4.660365104675293, "learning_rate": 9.063636363636365e-06, "num_tokens": 601880.0, "completions/mean_length": 238.5, "completions/min_length": 218.0, "completions/max_length": 279.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 238.5, "completions/min_terminated_length": 218.0, "completions/max_terminated_length": 279.0, "rewards/meter/mean": 0.9824122190475464, "rewards/meter/std": 0.030818087980151176, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.05892555043101311, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9860293865203857, "rewards/repeat_soft/std": 0.00635543605312705, "rewards/judge_quality/mean": 0.30124998092651367, "rewards/judge_quality/std": 0.10398317128419876, "rewards/total_composite/mean": 0.752938449382782, "rewards/total_composite/std": 0.03197839483618736, "reward": 0.752938449382782, "reward_std": 0.03197839483618736, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16094614565372467, "sampling/sampling_logp_difference/max": 1.2527246475219727, "sampling/importance_sampling_ratio/min": 0.2857252359390259, "sampling/importance_sampling_ratio/mean": 1.0237462520599365, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1439293324947357, "clip_ratio/low_mean": 0.0644422397017479, "clip_ratio/low_min": 0.0644422397017479, "clip_ratio/high_mean": 0.051997317001223564, "clip_ratio/high_max": 0.051997317001223564, "clip_ratio/region_mean": 0.11643955670297146, "reward_total_mean": 0.752938449382782, "reward_meter_mean": 0.9824122190475464, "reward_meter_std": 0.030818087980151176, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.05892555043101311, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9860293865203857, "reward_repeat_soft_std": 0.00635543605312705, "reward_judge_quality_mean": 0.30124998092651367, "reward_judge_quality_std": 0.10398317128419876, "reward_total_composite_mean": 0.752938449382782, "reward_total_composite_std": 0.03197839483618736} {"timestamp_utc": "2026-04-12T22:59:22Z", "mode": "train", "global_step": 311, "epoch": 0.03124058262179809, "loss": 0.1249, "grad_norm": 8.80500316619873, "learning_rate": 9.06060606060606e-06, "num_tokens": 604155.0, "completions/mean_length": 112.375, "completions/min_length": 71.0, "completions/max_length": 157.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.375, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.7527118921279907, "rewards/meter/std": 0.2170543670654297, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.972818911075592, "rewards/repeat_soft/std": 0.023975592106580734, "rewards/judge_quality/mean": 0.4024999737739563, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.7020647525787354, "rewards/total_composite/std": 0.11598742753267288, "reward": 0.7020647525787354, "reward_std": 0.11598742753267288, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16054801642894745, "sampling/sampling_logp_difference/max": 1.8488922119140625, "sampling/importance_sampling_ratio/min": 0.267759770154953, "sampling/importance_sampling_ratio/mean": 1.0379830598831177, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8777318000793457, "clip_ratio/low_mean": 0.0575484661385417, "clip_ratio/low_min": 0.0575484661385417, "clip_ratio/high_mean": 0.07582213822752237, "clip_ratio/high_max": 0.07582213822752237, "clip_ratio/region_mean": 0.13337060436606407, "reward_total_mean": 0.7020647525787354, "reward_meter_mean": 0.7527118921279907, "reward_meter_std": 0.2170543670654297, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.972818911075592, "reward_repeat_soft_std": 0.023975592106580734, "reward_judge_quality_mean": 0.4024999737739563, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.7020647525787354, "reward_total_composite_std": 0.11598742753267288} {"timestamp_utc": "2026-04-12T22:59:28Z", "mode": "train", "global_step": 312, "epoch": 0.03134103465595178, "loss": 0.0367, "grad_norm": 23.418533325195312, "learning_rate": 9.057575757575759e-06, "num_tokens": 605634.0, "completions/mean_length": 30.875, "completions/min_length": 28.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.875, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.6479890942573547, "rewards/meter/std": 0.37165024876594543, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.3174999952316284, "rewards/judge_quality/std": 0.0936177596449852, "rewards/total_composite/mean": 0.6330950856208801, "rewards/total_composite/std": 0.1794489473104477, "reward": 0.6330950856208801, "reward_std": 0.1794489473104477, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17153137922286987, "sampling/sampling_logp_difference/max": 1.188124179840088, "sampling/importance_sampling_ratio/min": 0.30479246377944946, "sampling/importance_sampling_ratio/mean": 1.018839955329895, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2423281371593475, "clip_ratio/low_mean": 0.0500576039776206, "clip_ratio/low_min": 0.0500576039776206, "clip_ratio/high_mean": 0.09362745378166437, "clip_ratio/high_max": 0.09362745378166437, "clip_ratio/region_mean": 0.14368505775928497, "reward_total_mean": 0.6330950856208801, "reward_meter_mean": 0.6479890942573547, "reward_meter_std": 0.37165024876594543, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.3174999952316284, "reward_judge_quality_std": 0.0936177596449852, "reward_total_composite_mean": 0.6330950856208801, "reward_total_composite_std": 0.1794489473104477} {"timestamp_utc": "2026-04-12T22:59:36Z", "mode": "train", "global_step": 313, "epoch": 0.03144148669010548, "loss": 0.0066, "grad_norm": 8.328590393066406, "learning_rate": 9.054545454545455e-06, "num_tokens": 608275.0, "completions/mean_length": 134.125, "completions/min_length": 105.0, "completions/max_length": 169.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 134.125, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 169.0, "rewards/meter/mean": 0.622697114944458, "rewards/meter/std": 0.300429105758667, "rewards/count_adherence/mean": 0.8500000238418579, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9915390014648438, "rewards/repeat_soft/std": 0.0050716460682451725, "rewards/judge_quality/mean": 0.39374998211860657, "rewards/judge_quality/std": 0.15638209879398346, "rewards/total_composite/mean": 0.5416899919509888, "rewards/total_composite/std": 0.2789454758167267, "reward": 0.5416899919509888, "reward_std": 0.2789454758167267, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21415948867797852, "sampling/sampling_logp_difference/max": 2.8406765460968018, "sampling/importance_sampling_ratio/min": 0.05838615074753761, "sampling/importance_sampling_ratio/mean": 1.025801420211792, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.665984809398651, "clip_ratio/low_mean": 0.08629223797470331, "clip_ratio/low_min": 0.08629223797470331, "clip_ratio/high_mean": 0.10891014710068703, "clip_ratio/high_max": 0.10891014710068703, "clip_ratio/region_mean": 0.19520238507539034, "reward_total_mean": 0.5416899919509888, "reward_meter_mean": 0.622697114944458, "reward_meter_std": 0.300429105758667, "reward_count_adherence_mean": 0.8500000238418579, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9915390014648438, "reward_repeat_soft_std": 0.0050716460682451725, "reward_judge_quality_mean": 0.39374998211860657, "reward_judge_quality_std": 0.15638209879398346, "reward_total_composite_mean": 0.5416899919509888, "reward_total_composite_std": 0.2789454758167267} {"timestamp_utc": "2026-04-12T22:59:42Z", "mode": "train", "global_step": 314, "epoch": 0.031541938724259165, "loss": 0.2793, "grad_norm": 18.40378189086914, "learning_rate": 9.051515151515152e-06, "num_tokens": 609673.0, "completions/mean_length": 38.75, "completions/min_length": 33.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9931943416595459, "rewards/meter/std": 0.005444410257041454, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.966524600982666, "rewards/repeat_soft/std": 0.011383358389139175, "rewards/judge_quality/mean": 0.39249998331069946, "rewards/judge_quality/std": 0.08892211318016052, "rewards/total_composite/mean": 0.7925899028778076, "rewards/total_composite/std": 0.05451717600226402, "reward": 0.7925899028778076, "reward_std": 0.05451717972755432, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1451827436685562, "sampling/sampling_logp_difference/max": 1.6814312934875488, "sampling/importance_sampling_ratio/min": 0.18610739707946777, "sampling/importance_sampling_ratio/mean": 1.0314937829971313, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5751785337924957, "clip_ratio/low_mean": 0.05078553315252066, "clip_ratio/low_min": 0.05078553315252066, "clip_ratio/high_mean": 0.12100310064852238, "clip_ratio/high_max": 0.12100310064852238, "clip_ratio/region_mean": 0.17178863380104303, "reward_total_mean": 0.7925899028778076, "reward_meter_mean": 0.9931943416595459, "reward_meter_std": 0.005444410257041454, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.966524600982666, "reward_repeat_soft_std": 0.011383358389139175, "reward_judge_quality_mean": 0.39249998331069946, "reward_judge_quality_std": 0.08892211318016052, "reward_total_composite_mean": 0.7925899028778076, "reward_total_composite_std": 0.05451717600226402} {"timestamp_utc": "2026-04-12T22:59:50Z", "mode": "train", "global_step": 315, "epoch": 0.03164239075841286, "loss": -0.0226, "grad_norm": 13.083516120910645, "learning_rate": 9.04848484848485e-06, "num_tokens": 611361.0, "completions/mean_length": 57.0, "completions/min_length": 48.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.0, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9411711096763611, "rewards/meter/std": 0.1282549947500229, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9782876968383789, "rewards/repeat_soft/std": 0.022529060021042824, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.8018558025360107, "rewards/total_composite/std": 0.061530958861112595, "reward": 0.8018558025360107, "reward_std": 0.061530955135822296, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15674841403961182, "sampling/sampling_logp_difference/max": 1.6053524017333984, "sampling/importance_sampling_ratio/min": 0.20081877708435059, "sampling/importance_sampling_ratio/mean": 1.0245052576065063, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3114160746335983, "clip_ratio/low_mean": 0.023574561811983585, "clip_ratio/low_min": 0.023574561811983585, "clip_ratio/high_mean": 0.10491535998880863, "clip_ratio/high_max": 0.10491535998880863, "clip_ratio/region_mean": 0.12848992180079222, "reward_total_mean": 0.8018558025360107, "reward_meter_mean": 0.9411711096763611, "reward_meter_std": 0.1282549947500229, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9782876968383789, "reward_repeat_soft_std": 0.022529060021042824, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.8018558025360107, "reward_total_composite_std": 0.061530958861112595} {"timestamp_utc": "2026-04-12T22:59:56Z", "mode": "train", "global_step": 316, "epoch": 0.03174284279256655, "loss": -0.1036, "grad_norm": 12.06344985961914, "learning_rate": 9.045454545454546e-06, "num_tokens": 613216.0, "completions/mean_length": 74.875, "completions/min_length": 49.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.875, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.43827885389328003, "rewards/meter/std": 0.4144741892814636, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9709675312042236, "rewards/repeat_soft/std": 0.0265809316188097, "rewards/judge_quality/mean": 0.47749999165534973, "rewards/judge_quality/std": 0.2303258627653122, "rewards/total_composite/mean": 0.5875722765922546, "rewards/total_composite/std": 0.14878006279468536, "reward": 0.5875722765922546, "reward_std": 0.14878006279468536, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1939167082309723, "sampling/sampling_logp_difference/max": 1.6496295928955078, "sampling/importance_sampling_ratio/min": 0.23769451677799225, "sampling/importance_sampling_ratio/mean": 1.0452182292938232, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7798522412776947, "clip_ratio/low_mean": 0.10747845284640789, "clip_ratio/low_min": 0.10747845284640789, "clip_ratio/high_mean": 0.08748633787035942, "clip_ratio/high_max": 0.08748633787035942, "clip_ratio/region_mean": 0.1949647907167673, "reward_total_mean": 0.5875722765922546, "reward_meter_mean": 0.43827885389328003, "reward_meter_std": 0.4144741892814636, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9709675312042236, "reward_repeat_soft_std": 0.0265809316188097, "reward_judge_quality_mean": 0.47749999165534973, "reward_judge_quality_std": 0.2303258627653122, "reward_total_composite_mean": 0.5875722765922546, "reward_total_composite_std": 0.14878006279468536} {"timestamp_utc": "2026-04-12T23:00:04Z", "mode": "train", "global_step": 317, "epoch": 0.03184329482672024, "loss": 0.0616, "grad_norm": 14.265595436096191, "learning_rate": 9.042424242424244e-06, "num_tokens": 614999.0, "completions/mean_length": 45.875, "completions/min_length": 32.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.875, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.7320458889007568, "rewards/meter/std": 0.25969332456588745, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9952741265296936, "rewards/repeat_soft/std": 0.0078080384992063046, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.1011011004447937, "rewards/total_composite/mean": 0.7206981182098389, "rewards/total_composite/std": 0.11563348770141602, "reward": 0.7206981182098389, "reward_std": 0.11563346534967422, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20902778208255768, "sampling/sampling_logp_difference/max": 1.2514324188232422, "sampling/importance_sampling_ratio/min": 0.28609469532966614, "sampling/importance_sampling_ratio/mean": 1.0482865571975708, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.311731696128845, "clip_ratio/low_mean": 0.09780640061944723, "clip_ratio/low_min": 0.09780640061944723, "clip_ratio/high_mean": 0.10708917211741209, "clip_ratio/high_max": 0.10708917211741209, "clip_ratio/region_mean": 0.20489557273685932, "reward_total_mean": 0.7206981182098389, "reward_meter_mean": 0.7320458889007568, "reward_meter_std": 0.25969332456588745, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9952741265296936, "reward_repeat_soft_std": 0.0078080384992063046, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.1011011004447937, "reward_total_composite_mean": 0.7206981182098389, "reward_total_composite_std": 0.11563348770141602} {"timestamp_utc": "2026-04-12T23:00:11Z", "mode": "train", "global_step": 318, "epoch": 0.03194374686087393, "loss": -0.0324, "grad_norm": 12.758769035339355, "learning_rate": 9.03939393939394e-06, "num_tokens": 617016.0, "completions/mean_length": 84.125, "completions/min_length": 62.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.125, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.6269983053207397, "rewards/meter/std": 0.24108316004276276, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9893389940261841, "rewards/repeat_soft/std": 0.008901750668883324, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.2243562489748001, "rewards/total_composite/mean": 0.5599960684776306, "rewards/total_composite/std": 0.25930678844451904, "reward": 0.5599960684776306, "reward_std": 0.25930678844451904, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21213707327842712, "sampling/sampling_logp_difference/max": 1.7732734680175781, "sampling/importance_sampling_ratio/min": 0.1697763204574585, "sampling/importance_sampling_ratio/mean": 1.025049090385437, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0486034005880356, "clip_ratio/low_mean": 0.09630942717194557, "clip_ratio/low_min": 0.09630942717194557, "clip_ratio/high_mean": 0.13631447590887547, "clip_ratio/high_max": 0.13631447590887547, "clip_ratio/region_mean": 0.23262390308082104, "reward_total_mean": 0.5599960684776306, "reward_meter_mean": 0.6269983053207397, "reward_meter_std": 0.24108316004276276, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9893389940261841, "reward_repeat_soft_std": 0.008901750668883324, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.2243562489748001, "reward_total_composite_mean": 0.5599960684776306, "reward_total_composite_std": 0.25930678844451904} {"timestamp_utc": "2026-04-12T23:00:17Z", "mode": "train", "global_step": 319, "epoch": 0.032044198895027624, "loss": 0.0163, "grad_norm": 15.600703239440918, "learning_rate": 9.036363636363638e-06, "num_tokens": 618485.0, "completions/mean_length": 33.625, "completions/min_length": 29.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.625, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9013193845748901, "rewards/meter/std": 0.18400028347969055, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9441953897476196, "rewards/repeat_soft/std": 0.034132927656173706, "rewards/judge_quality/mean": 0.45625001192092896, "rewards/judge_quality/std": 0.21185827255249023, "rewards/total_composite/mean": 0.7868882417678833, "rewards/total_composite/std": 0.10849127918481827, "reward": 0.7868882417678833, "reward_std": 0.10849127173423767, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1413993388414383, "sampling/sampling_logp_difference/max": 1.1563763618469238, "sampling/importance_sampling_ratio/min": 0.31462419033050537, "sampling/importance_sampling_ratio/mean": 1.011846661567688, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2068625912070274, "clip_ratio/low_mean": 0.03711166372522712, "clip_ratio/low_min": 0.03711166372522712, "clip_ratio/high_mean": 0.11634687706828117, "clip_ratio/high_max": 0.11634687706828117, "clip_ratio/region_mean": 0.1534585407935083, "reward_total_mean": 0.7868882417678833, "reward_meter_mean": 0.9013193845748901, "reward_meter_std": 0.18400028347969055, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9441953897476196, "reward_repeat_soft_std": 0.034132927656173706, "reward_judge_quality_mean": 0.45625001192092896, "reward_judge_quality_std": 0.21185827255249023, "reward_total_composite_mean": 0.7868882417678833, "reward_total_composite_std": 0.10849127918481827} {"timestamp_utc": "2026-04-12T23:00:23Z", "mode": "train", "global_step": 320, "epoch": 0.03214465092918132, "loss": 0.1484, "grad_norm": 24.80598258972168, "learning_rate": 9.033333333333334e-06, "num_tokens": 620062.0, "completions/mean_length": 35.125, "completions/min_length": 27.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.125, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.5973739624023438, "rewards/meter/std": 0.359290212392807, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9592273831367493, "rewards/repeat_soft/std": 0.09404433518648148, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.25150617957115173, "rewards/total_composite/mean": 0.6718660593032837, "rewards/total_composite/std": 0.2122698873281479, "reward": 0.6718660593032837, "reward_std": 0.2122698724269867, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17414693534374237, "sampling/sampling_logp_difference/max": 2.346524715423584, "sampling/importance_sampling_ratio/min": 0.0957011729478836, "sampling/importance_sampling_ratio/mean": 1.008365511894226, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9930548369884491, "clip_ratio/low_mean": 0.10787019226700068, "clip_ratio/low_min": 0.10787019226700068, "clip_ratio/high_mean": 0.06132630445063114, "clip_ratio/high_max": 0.06132630445063114, "clip_ratio/region_mean": 0.16919649671763182, "reward_total_mean": 0.6718660593032837, "reward_meter_mean": 0.5973739624023438, "reward_meter_std": 0.359290212392807, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9592273831367493, "reward_repeat_soft_std": 0.09404433518648148, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.25150617957115173, "reward_total_composite_mean": 0.6718660593032837, "reward_total_composite_std": 0.2122698873281479} {"timestamp_utc": "2026-04-12T23:00:30Z", "mode": "train", "global_step": 321, "epoch": 0.03224510296333501, "loss": -0.0311, "grad_norm": 11.990962028503418, "learning_rate": 9.030303030303031e-06, "num_tokens": 621991.0, "completions/mean_length": 79.125, "completions/min_length": 60.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.125, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.5917415618896484, "rewards/meter/std": 0.3370380401611328, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9915956258773804, "rewards/repeat_soft/std": 0.014456928707659245, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.6249432563781738, "rewards/total_composite/std": 0.15154770016670227, "reward": 0.6249432563781738, "reward_std": 0.15154768526554108, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2121298611164093, "sampling/sampling_logp_difference/max": 1.991551399230957, "sampling/importance_sampling_ratio/min": 0.13648352026939392, "sampling/importance_sampling_ratio/mean": 1.033273696899414, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.856611929833889, "clip_ratio/low_mean": 0.1014880994334817, "clip_ratio/low_min": 0.1014880994334817, "clip_ratio/high_mean": 0.08871226105839014, "clip_ratio/high_max": 0.08871226105839014, "clip_ratio/region_mean": 0.19020036049187183, "reward_total_mean": 0.6249432563781738, "reward_meter_mean": 0.5917415618896484, "reward_meter_std": 0.3370380401611328, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9915956258773804, "reward_repeat_soft_std": 0.014456928707659245, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.6249432563781738, "reward_total_composite_std": 0.15154770016670227} {"timestamp_utc": "2026-04-12T23:00:38Z", "mode": "train", "global_step": 322, "epoch": 0.0323455549974887, "loss": -0.1252, "grad_norm": 7.7470703125, "learning_rate": 9.027272727272728e-06, "num_tokens": 624312.0, "completions/mean_length": 127.125, "completions/min_length": 92.0, "completions/max_length": 180.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.125, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 180.0, "rewards/meter/mean": 0.6281640529632568, "rewards/meter/std": 0.35755035281181335, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9853676557540894, "rewards/repeat_soft/std": 0.010346058756113052, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.17266297340393066, "rewards/total_composite/mean": 0.5722471475601196, "rewards/total_composite/std": 0.28251969814300537, "reward": 0.5722471475601196, "reward_std": 0.28251972794532776, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20295292139053345, "sampling/sampling_logp_difference/max": 1.6357202529907227, "sampling/importance_sampling_ratio/min": 0.19481201469898224, "sampling/importance_sampling_ratio/mean": 1.0450093746185303, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.860815703868866, "clip_ratio/low_mean": 0.07746878080070019, "clip_ratio/low_min": 0.07746878080070019, "clip_ratio/high_mean": 0.11238695867359638, "clip_ratio/high_max": 0.11238695867359638, "clip_ratio/region_mean": 0.18985573947429657, "reward_total_mean": 0.5722471475601196, "reward_meter_mean": 0.6281640529632568, "reward_meter_std": 0.35755035281181335, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9853676557540894, "reward_repeat_soft_std": 0.010346058756113052, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.17266297340393066, "reward_total_composite_mean": 0.5722471475601196, "reward_total_composite_std": 0.28251969814300537} {"timestamp_utc": "2026-04-12T23:00:44Z", "mode": "train", "global_step": 323, "epoch": 0.03244600703164239, "loss": 0.1649, "grad_norm": 17.195138931274414, "learning_rate": 9.024242424242426e-06, "num_tokens": 626046.0, "completions/mean_length": 49.75, "completions/min_length": 32.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.75, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.8588954210281372, "rewards/meter/std": 0.34463977813720703, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9935814142227173, "rewards/repeat_soft/std": 0.007869238965213299, "rewards/judge_quality/mean": 0.46000000834465027, "rewards/judge_quality/std": 0.21138995885849, "rewards/total_composite/mean": 0.7738610506057739, "rewards/total_composite/std": 0.19971737265586853, "reward": 0.7738610506057739, "reward_std": 0.19971738755702972, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18340595066547394, "sampling/sampling_logp_difference/max": 1.7261734008789062, "sampling/importance_sampling_ratio/min": 0.17796410620212555, "sampling/importance_sampling_ratio/mean": 1.034714698791504, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6643513888120651, "clip_ratio/low_mean": 0.0234375, "clip_ratio/low_min": 0.0234375, "clip_ratio/high_mean": 0.1520734466612339, "clip_ratio/high_max": 0.1520734466612339, "clip_ratio/region_mean": 0.1755109466612339, "reward_total_mean": 0.7738610506057739, "reward_meter_mean": 0.8588954210281372, "reward_meter_std": 0.34463977813720703, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9935814142227173, "reward_repeat_soft_std": 0.007869238965213299, "reward_judge_quality_mean": 0.46000000834465027, "reward_judge_quality_std": 0.21138995885849, "reward_total_composite_mean": 0.7738610506057739, "reward_total_composite_std": 0.19971737265586853} {"timestamp_utc": "2026-04-12T23:00:51Z", "mode": "train", "global_step": 324, "epoch": 0.032546459065796084, "loss": 0.082, "grad_norm": 13.834494590759277, "learning_rate": 9.021212121212121e-06, "num_tokens": 627760.0, "completions/mean_length": 54.25, "completions/min_length": 46.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.25, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.5485538244247437, "rewards/meter/std": 0.3830815255641937, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9931224584579468, "rewards/repeat_soft/std": 0.007173935417085886, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.6232864856719971, "rewards/total_composite/std": 0.17412956058979034, "reward": 0.6232864856719971, "reward_std": 0.17412956058979034, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16963167488574982, "sampling/sampling_logp_difference/max": 1.6516146659851074, "sampling/importance_sampling_ratio/min": 0.19174006581306458, "sampling/importance_sampling_ratio/mean": 1.0151695013046265, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6320413798093796, "clip_ratio/low_mean": 0.07642136793583632, "clip_ratio/low_min": 0.07642136793583632, "clip_ratio/high_mean": 0.08271784521639347, "clip_ratio/high_max": 0.08271784521639347, "clip_ratio/region_mean": 0.15913921315222979, "reward_total_mean": 0.6232864856719971, "reward_meter_mean": 0.5485538244247437, "reward_meter_std": 0.3830815255641937, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9931224584579468, "reward_repeat_soft_std": 0.007173935417085886, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.6232864856719971, "reward_total_composite_std": 0.17412956058979034} {"timestamp_utc": "2026-04-12T23:00:58Z", "mode": "train", "global_step": 325, "epoch": 0.03264691109994977, "loss": 0.0226, "grad_norm": 11.458996772766113, "learning_rate": 9.01818181818182e-06, "num_tokens": 629549.0, "completions/mean_length": 60.625, "completions/min_length": 44.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.625, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.7409203052520752, "rewards/meter/std": 0.4324893653392792, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9784325957298279, "rewards/repeat_soft/std": 0.029624393209815025, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.2562922537326813, "rewards/total_composite/mean": 0.5041630864143372, "rewards/total_composite/std": 0.34416306018829346, "reward": 0.5041630864143372, "reward_std": 0.34416306018829346, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16835422813892365, "sampling/sampling_logp_difference/max": 1.6082210540771484, "sampling/importance_sampling_ratio/min": 0.2002435326576233, "sampling/importance_sampling_ratio/mean": 1.0246200561523438, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.652799054980278, "clip_ratio/low_mean": 0.05053866747766733, "clip_ratio/low_min": 0.05053866747766733, "clip_ratio/high_mean": 0.09803136438131332, "clip_ratio/high_max": 0.09803136438131332, "clip_ratio/region_mean": 0.14857003185898066, "reward_total_mean": 0.5041630864143372, "reward_meter_mean": 0.7409203052520752, "reward_meter_std": 0.4324893653392792, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9784325957298279, "reward_repeat_soft_std": 0.029624393209815025, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.2562922537326813, "reward_total_composite_mean": 0.5041630864143372, "reward_total_composite_std": 0.34416306018829346} {"timestamp_utc": "2026-04-12T23:01:05Z", "mode": "train", "global_step": 326, "epoch": 0.032747363134103466, "loss": 0.1143, "grad_norm": 14.518750190734863, "learning_rate": 9.015151515151516e-06, "num_tokens": 631093.0, "completions/mean_length": 49.0, "completions/min_length": 30.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.0, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.5105204582214355, "rewards/meter/std": 0.4025677442550659, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9925767183303833, "rewards/repeat_soft/std": 0.01736462116241455, "rewards/judge_quality/mean": 0.7362500429153442, "rewards/judge_quality/std": 0.262076199054718, "rewards/total_composite/mean": 0.6998668909072876, "rewards/total_composite/std": 0.2293824404478073, "reward": 0.6998668909072876, "reward_std": 0.22938242554664612, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.168900266289711, "sampling/sampling_logp_difference/max": 1.8604044914245605, "sampling/importance_sampling_ratio/min": 0.15560968220233917, "sampling/importance_sampling_ratio/mean": 1.0194494724273682, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.504062045365572, "clip_ratio/low_mean": 0.07193658966571093, "clip_ratio/low_min": 0.07193658966571093, "clip_ratio/high_mean": 0.07654862012714148, "clip_ratio/high_max": 0.07654862012714148, "clip_ratio/region_mean": 0.1484852097928524, "reward_total_mean": 0.6998668909072876, "reward_meter_mean": 0.5105204582214355, "reward_meter_std": 0.4025677442550659, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9925767183303833, "reward_repeat_soft_std": 0.01736462116241455, "reward_judge_quality_mean": 0.7362500429153442, "reward_judge_quality_std": 0.262076199054718, "reward_total_composite_mean": 0.6998668909072876, "reward_total_composite_std": 0.2293824404478073} {"timestamp_utc": "2026-04-12T23:01:12Z", "mode": "train", "global_step": 327, "epoch": 0.03284781516825716, "loss": -0.0117, "grad_norm": 14.595277786254883, "learning_rate": 9.012121212121213e-06, "num_tokens": 632488.0, "completions/mean_length": 38.375, "completions/min_length": 27.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.375, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.3120831251144409, "rewards/meter/std": 0.4151588976383209, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9965986013412476, "rewards/repeat_soft/std": 0.006567451637238264, "rewards/judge_quality/mean": 0.65625, "rewards/judge_quality/std": 0.2888864278793335, "rewards/total_composite/mean": 0.49166226387023926, "rewards/total_composite/std": 0.26445022225379944, "reward": 0.49166226387023926, "reward_std": 0.26445022225379944, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21693284809589386, "sampling/sampling_logp_difference/max": 2.0229439735412598, "sampling/importance_sampling_ratio/min": 0.13226550817489624, "sampling/importance_sampling_ratio/mean": 1.0046648979187012, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3793605118989944, "clip_ratio/low_mean": 0.10128630138933659, "clip_ratio/low_min": 0.10128630138933659, "clip_ratio/high_mean": 0.09768502600491047, "clip_ratio/high_max": 0.09768502600491047, "clip_ratio/region_mean": 0.19897132739424706, "reward_total_mean": 0.49166226387023926, "reward_meter_mean": 0.3120831251144409, "reward_meter_std": 0.4151588976383209, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9965986013412476, "reward_repeat_soft_std": 0.006567451637238264, "reward_judge_quality_mean": 0.65625, "reward_judge_quality_std": 0.2888864278793335, "reward_total_composite_mean": 0.49166226387023926, "reward_total_composite_std": 0.26445022225379944} {"timestamp_utc": "2026-04-12T23:01:17Z", "mode": "train", "global_step": 328, "epoch": 0.03294826720241085, "loss": 0.0527, "grad_norm": 20.85543441772461, "learning_rate": 9.00909090909091e-06, "num_tokens": 633981.0, "completions/mean_length": 27.625, "completions/min_length": 24.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.625, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.5585148334503174, "rewards/meter/std": 0.41181111335754395, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9621384143829346, "rewards/repeat_soft/std": 0.0010226722806692123, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.263435423374176, "rewards/total_composite/mean": 0.6996704936027527, "rewards/total_composite/std": 0.1905549019575119, "reward": 0.6996704936027527, "reward_std": 0.19055487215518951, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19206155836582184, "sampling/sampling_logp_difference/max": 1.153684139251709, "sampling/importance_sampling_ratio/min": 0.31547239422798157, "sampling/importance_sampling_ratio/mean": 1.0439910888671875, "sampling/importance_sampling_ratio/max": 1.8983124494552612, "entropy": 1.6457576304674149, "clip_ratio/low_mean": 0.0951724136248231, "clip_ratio/low_min": 0.0951724136248231, "clip_ratio/high_mean": 0.10912698600441217, "clip_ratio/high_max": 0.10912698600441217, "clip_ratio/region_mean": 0.20429939962923527, "reward_total_mean": 0.6996704936027527, "reward_meter_mean": 0.5585148334503174, "reward_meter_std": 0.41181111335754395, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9621384143829346, "reward_repeat_soft_std": 0.0010226722806692123, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.263435423374176, "reward_total_composite_mean": 0.6996704936027527, "reward_total_composite_std": 0.1905549019575119} {"timestamp_utc": "2026-04-12T23:01:24Z", "mode": "train", "global_step": 329, "epoch": 0.03304871923656454, "loss": -0.0124, "grad_norm": 15.625420570373535, "learning_rate": 9.006060606060607e-06, "num_tokens": 635641.0, "completions/mean_length": 32.5, "completions/min_length": 28.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.5, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.8752278685569763, "rewards/meter/std": 0.25363320112228394, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9494680762290955, "rewards/repeat_soft/std": 0.03246039152145386, "rewards/judge_quality/mean": 0.3050000071525574, "rewards/judge_quality/std": 0.10928338021039963, "rewards/total_composite/mean": 0.7302993535995483, "rewards/total_composite/std": 0.10676144808530807, "reward": 0.7302993535995483, "reward_std": 0.10676144808530807, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16113220155239105, "sampling/sampling_logp_difference/max": 1.450995922088623, "sampling/importance_sampling_ratio/min": 0.23433680832386017, "sampling/importance_sampling_ratio/mean": 1.039710521697998, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5074107348918915, "clip_ratio/low_mean": 0.036830357275903225, "clip_ratio/low_min": 0.036830357275903225, "clip_ratio/high_mean": 0.1013183337636292, "clip_ratio/high_max": 0.1013183337636292, "clip_ratio/region_mean": 0.13814869103953242, "reward_total_mean": 0.7302993535995483, "reward_meter_mean": 0.8752278685569763, "reward_meter_std": 0.25363320112228394, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9494680762290955, "reward_repeat_soft_std": 0.03246039152145386, "reward_judge_quality_mean": 0.3050000071525574, "reward_judge_quality_std": 0.10928338021039963, "reward_total_composite_mean": 0.7302993535995483, "reward_total_composite_std": 0.10676144808530807} {"timestamp_utc": "2026-04-12T23:01:30Z", "mode": "train", "global_step": 330, "epoch": 0.03314917127071823, "loss": 0.0388, "grad_norm": 18.413442611694336, "learning_rate": 9.003030303030303e-06, "num_tokens": 637312.0, "completions/mean_length": 38.875, "completions/min_length": 35.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.875, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.5271964073181152, "rewards/meter/std": 0.3047308921813965, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9943450093269348, "rewards/repeat_soft/std": 0.004511907696723938, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.617547869682312, "rewards/total_composite/std": 0.14538224041461945, "reward": 0.617547869682312, "reward_std": 0.14538224041461945, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1876116842031479, "sampling/sampling_logp_difference/max": 1.6785035133361816, "sampling/importance_sampling_ratio/min": 0.18665309250354767, "sampling/importance_sampling_ratio/mean": 1.009373426437378, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5645318925380707, "clip_ratio/low_mean": 0.06732303369790316, "clip_ratio/low_min": 0.06732303369790316, "clip_ratio/high_mean": 0.1049501858651638, "clip_ratio/high_max": 0.1049501858651638, "clip_ratio/region_mean": 0.17227321956306696, "reward_total_mean": 0.617547869682312, "reward_meter_mean": 0.5271964073181152, "reward_meter_std": 0.3047308921813965, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9943450093269348, "reward_repeat_soft_std": 0.004511907696723938, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.617547869682312, "reward_total_composite_std": 0.14538224041461945} {"timestamp_utc": "2026-04-12T23:01:36Z", "mode": "train", "global_step": 331, "epoch": 0.033249623304871925, "loss": 0.069, "grad_norm": 14.670364379882812, "learning_rate": 9e-06, "num_tokens": 638932.0, "completions/mean_length": 46.5, "completions/min_length": 26.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.5, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.541580319404602, "rewards/meter/std": 0.350299596786499, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9796052575111389, "rewards/repeat_soft/std": 0.021732060238718987, "rewards/judge_quality/mean": 0.5862500071525574, "rewards/judge_quality/std": 0.22984080016613007, "rewards/total_composite/mean": 0.6675466299057007, "rewards/total_composite/std": 0.19260407984256744, "reward": 0.6675466299057007, "reward_std": 0.19260407984256744, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1973562091588974, "sampling/sampling_logp_difference/max": 2.4663381576538086, "sampling/importance_sampling_ratio/min": 0.08489516377449036, "sampling/importance_sampling_ratio/mean": 1.017340898513794, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6219275891780853, "clip_ratio/low_mean": 0.0712700542062521, "clip_ratio/low_min": 0.0712700542062521, "clip_ratio/high_mean": 0.1228944119066, "clip_ratio/high_max": 0.1228944119066, "clip_ratio/region_mean": 0.1941644661128521, "reward_total_mean": 0.6675466299057007, "reward_meter_mean": 0.541580319404602, "reward_meter_std": 0.350299596786499, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9796052575111389, "reward_repeat_soft_std": 0.021732060238718987, "reward_judge_quality_mean": 0.5862500071525574, "reward_judge_quality_std": 0.22984080016613007, "reward_total_composite_mean": 0.6675466299057007, "reward_total_composite_std": 0.19260407984256744} {"timestamp_utc": "2026-04-12T23:01:42Z", "mode": "train", "global_step": 332, "epoch": 0.03335007533902561, "loss": 0.0757, "grad_norm": 19.00959014892578, "learning_rate": 8.996969696969697e-06, "num_tokens": 640632.0, "completions/mean_length": 44.5, "completions/min_length": 38.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.5, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.5692949891090393, "rewards/meter/std": 0.3381059467792511, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9929022789001465, "rewards/repeat_soft/std": 0.010078132152557373, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.19078318774700165, "rewards/total_composite/mean": 0.6625979542732239, "rewards/total_composite/std": 0.13137099146842957, "reward": 0.6625979542732239, "reward_std": 0.13137100636959076, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1934403032064438, "sampling/sampling_logp_difference/max": 1.4581849575042725, "sampling/importance_sampling_ratio/min": 0.23265817761421204, "sampling/importance_sampling_ratio/mean": 1.039157509803772, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.844116635620594, "clip_ratio/low_mean": 0.09967726096510887, "clip_ratio/low_min": 0.09967726096510887, "clip_ratio/high_mean": 0.06361806951463223, "clip_ratio/high_max": 0.06361806951463223, "clip_ratio/region_mean": 0.1632953304797411, "reward_total_mean": 0.6625979542732239, "reward_meter_mean": 0.5692949891090393, "reward_meter_std": 0.3381059467792511, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9929022789001465, "reward_repeat_soft_std": 0.010078132152557373, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.19078318774700165, "reward_total_composite_mean": 0.6625979542732239, "reward_total_composite_std": 0.13137099146842957} {"timestamp_utc": "2026-04-12T23:01:53Z", "mode": "train", "global_step": 333, "epoch": 0.03345052737317931, "loss": -0.1805, "grad_norm": 2.1921451091766357, "learning_rate": 8.993939393939395e-06, "num_tokens": 642960.0, "completions/mean_length": 169.0, "completions/min_length": 76.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 120.00000762939453, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 172.0, "rewards/meter/mean": 0.8693485260009766, "rewards/meter/std": 0.2024930864572525, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.993369460105896, "rewards/repeat_soft/std": 0.007170869503170252, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.6636952757835388, "rewards/total_composite/std": 0.27916404604911804, "reward": 0.6636952757835388, "reward_std": 0.27916401624679565, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1852688044309616, "sampling/sampling_logp_difference/max": 2.4294490814208984, "sampling/importance_sampling_ratio/min": 0.0880853459239006, "sampling/importance_sampling_ratio/mean": 1.036986231803894, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.442813605070114, "clip_ratio/low_mean": 0.012121211737394333, "clip_ratio/low_min": 0.012121211737394333, "clip_ratio/high_mean": 0.1361191812902689, "clip_ratio/high_max": 0.1361191812902689, "clip_ratio/region_mean": 0.14824039302766323, "reward_total_mean": 0.6636952757835388, "reward_meter_mean": 0.8693485260009766, "reward_meter_std": 0.2024930864572525, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.993369460105896, "reward_repeat_soft_std": 0.007170869503170252, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.6636952757835388, "reward_total_composite_std": 0.27916404604911804} {"timestamp_utc": "2026-04-12T23:02:01Z", "mode": "train", "global_step": 334, "epoch": 0.033550979407332995, "loss": -0.1087, "grad_norm": 9.211053848266602, "learning_rate": 8.990909090909092e-06, "num_tokens": 645171.0, "completions/mean_length": 107.375, "completions/min_length": 66.0, "completions/max_length": 141.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.375, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 141.0, "rewards/meter/mean": 0.5678161382675171, "rewards/meter/std": 0.2947660982608795, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9880356192588806, "rewards/repeat_soft/std": 0.015491855330765247, "rewards/judge_quality/mean": 0.39374998211860657, "rewards/judge_quality/std": 0.15638209879398346, "rewards/total_composite/mean": 0.5590454339981079, "rewards/total_composite/std": 0.2564302980899811, "reward": 0.5590454339981079, "reward_std": 0.2564302980899811, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1881706416606903, "sampling/sampling_logp_difference/max": 1.4340581893920898, "sampling/importance_sampling_ratio/min": 0.23833972215652466, "sampling/importance_sampling_ratio/mean": 1.045169472694397, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.440303310751915, "clip_ratio/low_mean": 0.06710651330649853, "clip_ratio/low_min": 0.06710651330649853, "clip_ratio/high_mean": 0.07854034472256899, "clip_ratio/high_max": 0.07854034472256899, "clip_ratio/region_mean": 0.14564685802906752, "reward_total_mean": 0.5590454339981079, "reward_meter_mean": 0.5678161382675171, "reward_meter_std": 0.2947660982608795, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9880356192588806, "reward_repeat_soft_std": 0.015491855330765247, "reward_judge_quality_mean": 0.39374998211860657, "reward_judge_quality_std": 0.15638209879398346, "reward_total_composite_mean": 0.5590454339981079, "reward_total_composite_std": 0.2564302980899811} {"timestamp_utc": "2026-04-12T23:02:08Z", "mode": "train", "global_step": 335, "epoch": 0.03365143144148669, "loss": 0.2411, "grad_norm": 13.905549049377441, "learning_rate": 8.98787878787879e-06, "num_tokens": 647321.0, "completions/mean_length": 83.75, "completions/min_length": 56.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.75, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.7348113059997559, "rewards/meter/std": 0.29187849164009094, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9888418912887573, "rewards/repeat_soft/std": 0.006775399204343557, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.7430492639541626, "rewards/total_composite/std": 0.14526678621768951, "reward": 0.7430492639541626, "reward_std": 0.14526678621768951, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17516358196735382, "sampling/sampling_logp_difference/max": 2.149899482727051, "sampling/importance_sampling_ratio/min": 0.11649586260318756, "sampling/importance_sampling_ratio/mean": 1.0206749439239502, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0544252693653107, "clip_ratio/low_mean": 0.050727492198348045, "clip_ratio/low_min": 0.050727492198348045, "clip_ratio/high_mean": 0.09651539474725723, "clip_ratio/high_max": 0.09651539474725723, "clip_ratio/region_mean": 0.14724288694560528, "reward_total_mean": 0.7430492639541626, "reward_meter_mean": 0.7348113059997559, "reward_meter_std": 0.29187849164009094, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9888418912887573, "reward_repeat_soft_std": 0.006775399204343557, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.7430492639541626, "reward_total_composite_std": 0.14526678621768951} {"timestamp_utc": "2026-04-12T23:02:14Z", "mode": "train", "global_step": 336, "epoch": 0.033751883475640385, "loss": 0.007, "grad_norm": 22.59181022644043, "learning_rate": 8.984848484848485e-06, "num_tokens": 648824.0, "completions/mean_length": 31.875, "completions/min_length": 24.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.875, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.8885046243667603, "rewards/meter/std": 0.1540706604719162, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9856975674629211, "rewards/repeat_soft/std": 0.012106251902878284, "rewards/judge_quality/mean": 0.9200000166893005, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.9243968725204468, "rewards/total_composite/std": 0.06845267862081528, "reward": 0.9243968725204468, "reward_std": 0.06845267117023468, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10955523699522018, "sampling/sampling_logp_difference/max": 2.364337682723999, "sampling/importance_sampling_ratio/min": 0.09401154518127441, "sampling/importance_sampling_ratio/mean": 1.009216070175171, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3893982842564583, "clip_ratio/low_mean": 0.03318452415987849, "clip_ratio/low_min": 0.03318452415987849, "clip_ratio/high_mean": 0.04626510385423899, "clip_ratio/high_max": 0.04626510385423899, "clip_ratio/region_mean": 0.07944962801411748, "reward_total_mean": 0.9243968725204468, "reward_meter_mean": 0.8885046243667603, "reward_meter_std": 0.1540706604719162, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9856975674629211, "reward_repeat_soft_std": 0.012106251902878284, "reward_judge_quality_mean": 0.9200000166893005, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.9243968725204468, "reward_total_composite_std": 0.06845267862081528} {"timestamp_utc": "2026-04-12T23:02:21Z", "mode": "train", "global_step": 337, "epoch": 0.03385233550979407, "loss": 0.0394, "grad_norm": 6.496931552886963, "learning_rate": 8.981818181818182e-06, "num_tokens": 651596.0, "completions/mean_length": 146.5, "completions/min_length": 132.0, "completions/max_length": 162.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 146.5, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 162.0, "rewards/meter/mean": 0.5146921873092651, "rewards/meter/std": 0.22382019460201263, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9559527635574341, "rewards/repeat_soft/std": 0.02431439608335495, "rewards/judge_quality/mean": 0.2762500047683716, "rewards/judge_quality/std": 0.12603145837783813, "rewards/total_composite/mean": 0.5300817489624023, "rewards/total_composite/std": 0.13059987127780914, "reward": 0.5300817489624023, "reward_std": 0.13059987127780914, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16727757453918457, "sampling/sampling_logp_difference/max": 1.865015983581543, "sampling/importance_sampling_ratio/min": 0.15489374101161957, "sampling/importance_sampling_ratio/mean": 1.0261808633804321, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6655433475971222, "clip_ratio/low_mean": 0.06393730640411377, "clip_ratio/low_min": 0.06393730640411377, "clip_ratio/high_mean": 0.09135236963629723, "clip_ratio/high_max": 0.09135236963629723, "clip_ratio/region_mean": 0.155289676040411, "reward_total_mean": 0.5300817489624023, "reward_meter_mean": 0.5146921873092651, "reward_meter_std": 0.22382019460201263, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9559527635574341, "reward_repeat_soft_std": 0.02431439608335495, "reward_judge_quality_mean": 0.2762500047683716, "reward_judge_quality_std": 0.12603145837783813, "reward_total_composite_mean": 0.5300817489624023, "reward_total_composite_std": 0.13059987127780914} {"timestamp_utc": "2026-04-12T23:02:28Z", "mode": "train", "global_step": 338, "epoch": 0.03395278754394777, "loss": 0.1072, "grad_norm": 9.586305618286133, "learning_rate": 8.97878787878788e-06, "num_tokens": 653648.0, "completions/mean_length": 85.5, "completions/min_length": 51.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.5, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.900834858417511, "rewards/meter/std": 0.23851878941059113, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.987109899520874, "rewards/repeat_soft/std": 0.012488877400755882, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.08166787773370743, "rewards/total_composite/mean": 0.7559616565704346, "rewards/total_composite/std": 0.12291630357503891, "reward": 0.7559616565704346, "reward_std": 0.1229163110256195, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1669662594795227, "sampling/sampling_logp_difference/max": 1.2253317832946777, "sampling/importance_sampling_ratio/min": 0.2936602532863617, "sampling/importance_sampling_ratio/mean": 1.0407122373580933, "sampling/importance_sampling_ratio/max": 1.8260409832000732, "entropy": 2.3771502524614334, "clip_ratio/low_mean": 0.024514411576092243, "clip_ratio/low_min": 0.024514411576092243, "clip_ratio/high_mean": 0.10875153262168169, "clip_ratio/high_max": 0.10875153262168169, "clip_ratio/region_mean": 0.13326594419777393, "reward_total_mean": 0.7559616565704346, "reward_meter_mean": 0.900834858417511, "reward_meter_std": 0.23851878941059113, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.987109899520874, "reward_repeat_soft_std": 0.012488877400755882, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.08166787773370743, "reward_total_composite_mean": 0.7559616565704346, "reward_total_composite_std": 0.12291630357503891} {"timestamp_utc": "2026-04-12T23:02:41Z", "mode": "train", "global_step": 339, "epoch": 0.034053239578101455, "loss": 0.0132, "grad_norm": 13.859871864318848, "learning_rate": 8.975757575757577e-06, "num_tokens": 655349.0, "completions/mean_length": 52.625, "completions/min_length": 34.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.625, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.7013536691665649, "rewards/meter/std": 0.3321683704853058, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9941227436065674, "rewards/repeat_soft/std": 0.011110487394034863, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.6880214214324951, "rewards/total_composite/std": 0.14885766804218292, "reward": 0.6880214214324951, "reward_std": 0.14885766804218292, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17653164267539978, "sampling/sampling_logp_difference/max": 1.4170807600021362, "sampling/importance_sampling_ratio/min": 0.24242067337036133, "sampling/importance_sampling_ratio/mean": 1.0118991136550903, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6389329507946968, "clip_ratio/low_mean": 0.08546487893909216, "clip_ratio/low_min": 0.08546487893909216, "clip_ratio/high_mean": 0.07128393277525902, "clip_ratio/high_max": 0.07128393277525902, "clip_ratio/region_mean": 0.15674881171435118, "reward_total_mean": 0.6880214214324951, "reward_meter_mean": 0.7013536691665649, "reward_meter_std": 0.3321683704853058, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9941227436065674, "reward_repeat_soft_std": 0.011110487394034863, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.6880214214324951, "reward_total_composite_std": 0.14885766804218292} {"timestamp_utc": "2026-04-12T23:02:47Z", "mode": "train", "global_step": 340, "epoch": 0.03415369161225515, "loss": -0.0541, "grad_norm": 11.287980079650879, "learning_rate": 8.972727272727272e-06, "num_tokens": 656913.0, "completions/mean_length": 46.5, "completions/min_length": 36.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9014050960540771, "rewards/meter/std": 0.24767223000526428, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9928537011146545, "rewards/repeat_soft/std": 0.012313371524214745, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.14201988279819489, "rewards/total_composite/mean": 0.8067926168441772, "rewards/total_composite/std": 0.1277669370174408, "reward": 0.8067926168441772, "reward_std": 0.1277669370174408, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1463606357574463, "sampling/sampling_logp_difference/max": 1.3862800598144531, "sampling/importance_sampling_ratio/min": 0.2500035762786865, "sampling/importance_sampling_ratio/mean": 1.0395745038986206, "sampling/importance_sampling_ratio/max": 1.785520076751709, "entropy": 1.8958442583680153, "clip_ratio/low_mean": 0.016447369009256363, "clip_ratio/low_min": 0.016447369009256363, "clip_ratio/high_mean": 0.17040428798645735, "clip_ratio/high_max": 0.17040428798645735, "clip_ratio/region_mean": 0.1868516569957137, "reward_total_mean": 0.8067926168441772, "reward_meter_mean": 0.9014050960540771, "reward_meter_std": 0.24767223000526428, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9928537011146545, "reward_repeat_soft_std": 0.012313371524214745, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.14201988279819489, "reward_total_composite_mean": 0.8067926168441772, "reward_total_composite_std": 0.1277669370174408} {"timestamp_utc": "2026-04-12T23:02:54Z", "mode": "train", "global_step": 341, "epoch": 0.03425414364640884, "loss": 0.1788, "grad_norm": 12.435867309570312, "learning_rate": 8.969696969696971e-06, "num_tokens": 658880.0, "completions/mean_length": 71.875, "completions/min_length": 48.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.875, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.45456400513648987, "rewards/meter/std": 0.2630288898944855, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9900200366973877, "rewards/repeat_soft/std": 0.016991987824440002, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.5478057861328125, "rewards/total_composite/std": 0.12569832801818848, "reward": 0.5478057861328125, "reward_std": 0.12569832801818848, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20303896069526672, "sampling/sampling_logp_difference/max": 2.090424060821533, "sampling/importance_sampling_ratio/min": 0.1236346960067749, "sampling/importance_sampling_ratio/mean": 1.0307660102844238, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6375721842050552, "clip_ratio/low_mean": 0.05487620737403631, "clip_ratio/low_min": 0.05487620737403631, "clip_ratio/high_mean": 0.14256374537944794, "clip_ratio/high_max": 0.14256374537944794, "clip_ratio/region_mean": 0.19743995275348425, "reward_total_mean": 0.5478057861328125, "reward_meter_mean": 0.45456400513648987, "reward_meter_std": 0.2630288898944855, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9900200366973877, "reward_repeat_soft_std": 0.016991987824440002, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.5478057861328125, "reward_total_composite_std": 0.12569832801818848} {"timestamp_utc": "2026-04-12T23:03:05Z", "mode": "train", "global_step": 342, "epoch": 0.03435459568056253, "loss": -0.1406, "grad_norm": 3.093289852142334, "learning_rate": 8.966666666666667e-06, "num_tokens": 661206.0, "completions/mean_length": 238.75, "completions/min_length": 129.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 147.6666717529297, "completions/min_terminated_length": 129.0, "completions/max_terminated_length": 161.0, "rewards/meter/mean": 0.54021155834198, "rewards/meter/std": 0.32083621621131897, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9923524856567383, "rewards/repeat_soft/std": 0.009116299450397491, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.27166154980659485, "rewards/total_composite/mean": 0.37448006868362427, "rewards/total_composite/std": 0.35103604197502136, "reward": 0.37448006868362427, "reward_std": 0.35103604197502136, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16647261381149292, "sampling/sampling_logp_difference/max": 4.441570281982422, "sampling/importance_sampling_ratio/min": 0.011777430772781372, "sampling/importance_sampling_ratio/mean": 1.018772840499878, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5200306177139282, "clip_ratio/low_mean": 0.01953125, "clip_ratio/low_min": 0.01953125, "clip_ratio/high_mean": 0.09901874419301748, "clip_ratio/high_max": 0.09901874419301748, "clip_ratio/region_mean": 0.11854999419301748, "reward_total_mean": 0.37448006868362427, "reward_meter_mean": 0.54021155834198, "reward_meter_std": 0.32083621621131897, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9923524856567383, "reward_repeat_soft_std": 0.009116299450397491, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.27166154980659485, "reward_total_composite_mean": 0.37448006868362427, "reward_total_composite_std": 0.35103604197502136} {"timestamp_utc": "2026-04-12T23:03:11Z", "mode": "train", "global_step": 343, "epoch": 0.034455047714716226, "loss": 0.2887, "grad_norm": 19.69594955444336, "learning_rate": 8.963636363636364e-06, "num_tokens": 662669.0, "completions/mean_length": 28.875, "completions/min_length": 18.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.875, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.5758342742919922, "rewards/meter/std": 0.3773238956928253, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9179693460464478, "rewards/repeat_soft/std": 0.0841730460524559, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.24833375215530396, "rewards/total_composite/mean": 0.6415473222732544, "rewards/total_composite/std": 0.20620712637901306, "reward": 0.6415473222732544, "reward_std": 0.20620709657669067, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1966894567012787, "sampling/sampling_logp_difference/max": 1.3404908180236816, "sampling/importance_sampling_ratio/min": 0.26171717047691345, "sampling/importance_sampling_ratio/mean": 1.0354410409927368, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0854700058698654, "clip_ratio/low_mean": 0.08456349186599255, "clip_ratio/low_min": 0.08456349186599255, "clip_ratio/high_mean": 0.08873996417969465, "clip_ratio/high_max": 0.08873996417969465, "clip_ratio/region_mean": 0.1733034560456872, "reward_total_mean": 0.6415473222732544, "reward_meter_mean": 0.5758342742919922, "reward_meter_std": 0.3773238956928253, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9179693460464478, "reward_repeat_soft_std": 0.0841730460524559, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.24833375215530396, "reward_total_composite_mean": 0.6415473222732544, "reward_total_composite_std": 0.20620712637901306} {"timestamp_utc": "2026-04-12T23:03:17Z", "mode": "train", "global_step": 344, "epoch": 0.034555499748869914, "loss": -0.0212, "grad_norm": 14.172499656677246, "learning_rate": 8.960606060606061e-06, "num_tokens": 664312.0, "completions/mean_length": 40.375, "completions/min_length": 29.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.375, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.637878954410553, "rewards/meter/std": 0.47764691710472107, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9966552257537842, "rewards/repeat_soft/std": 0.005645620170980692, "rewards/judge_quality/mean": 0.5425000190734863, "rewards/judge_quality/std": 0.18645183742046356, "rewards/total_composite/mean": 0.699461042881012, "rewards/total_composite/std": 0.20450200140476227, "reward": 0.699461042881012, "reward_std": 0.20450198650360107, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18251337110996246, "sampling/sampling_logp_difference/max": 1.6479954719543457, "sampling/importance_sampling_ratio/min": 0.19243526458740234, "sampling/importance_sampling_ratio/mean": 1.0123041868209839, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.494774580001831, "clip_ratio/low_mean": 0.07349334564059973, "clip_ratio/low_min": 0.07349334564059973, "clip_ratio/high_mean": 0.12716410867869854, "clip_ratio/high_max": 0.12716410867869854, "clip_ratio/region_mean": 0.20065745431929827, "reward_total_mean": 0.699461042881012, "reward_meter_mean": 0.637878954410553, "reward_meter_std": 0.47764691710472107, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9966552257537842, "reward_repeat_soft_std": 0.005645620170980692, "reward_judge_quality_mean": 0.5425000190734863, "reward_judge_quality_std": 0.18645183742046356, "reward_total_composite_mean": 0.699461042881012, "reward_total_composite_std": 0.20450200140476227} {"timestamp_utc": "2026-04-12T23:03:24Z", "mode": "train", "global_step": 345, "epoch": 0.03465595178302361, "loss": 0.0726, "grad_norm": 15.740530014038086, "learning_rate": 8.957575757575758e-06, "num_tokens": 665963.0, "completions/mean_length": 49.375, "completions/min_length": 34.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.375, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.5705506205558777, "rewards/meter/std": 0.3587588667869568, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9831625819206238, "rewards/repeat_soft/std": 0.02472248114645481, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.6340640187263489, "rewards/total_composite/std": 0.15570303797721863, "reward": 0.6340640187263489, "reward_std": 0.15570305287837982, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19708193838596344, "sampling/sampling_logp_difference/max": 2.8270010948181152, "sampling/importance_sampling_ratio/min": 0.059190090745687485, "sampling/importance_sampling_ratio/mean": 1.050620198249817, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1630947589874268, "clip_ratio/low_mean": 0.07492022961378098, "clip_ratio/low_min": 0.07492022961378098, "clip_ratio/high_mean": 0.1355973221361637, "clip_ratio/high_max": 0.1355973221361637, "clip_ratio/region_mean": 0.2105175517499447, "reward_total_mean": 0.6340640187263489, "reward_meter_mean": 0.5705506205558777, "reward_meter_std": 0.3587588667869568, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9831625819206238, "reward_repeat_soft_std": 0.02472248114645481, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.6340640187263489, "reward_total_composite_std": 0.15570303797721863} {"timestamp_utc": "2026-04-12T23:03:31Z", "mode": "train", "global_step": 346, "epoch": 0.034756403817177296, "loss": 0.0613, "grad_norm": 9.681682586669922, "learning_rate": 8.954545454545456e-06, "num_tokens": 667881.0, "completions/mean_length": 85.75, "completions/min_length": 54.0, "completions/max_length": 119.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.75, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.7363324165344238, "rewards/meter/std": 0.3174753487110138, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9872612953186035, "rewards/repeat_soft/std": 0.010840080678462982, "rewards/judge_quality/mean": 0.44875001907348633, "rewards/judge_quality/std": 0.2125651240348816, "rewards/total_composite/mean": 0.7084506750106812, "rewards/total_composite/std": 0.12094394117593765, "reward": 0.7084506750106812, "reward_std": 0.12094393372535706, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.164615198969841, "sampling/sampling_logp_difference/max": 1.8680634498596191, "sampling/importance_sampling_ratio/min": 0.15442241728305817, "sampling/importance_sampling_ratio/mean": 1.013057827949524, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7178567796945572, "clip_ratio/low_mean": 0.05927484482526779, "clip_ratio/low_min": 0.05927484482526779, "clip_ratio/high_mean": 0.09947736468166113, "clip_ratio/high_max": 0.09947736468166113, "clip_ratio/region_mean": 0.15875220950692892, "reward_total_mean": 0.7084506750106812, "reward_meter_mean": 0.7363324165344238, "reward_meter_std": 0.3174753487110138, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9872612953186035, "reward_repeat_soft_std": 0.010840080678462982, "reward_judge_quality_mean": 0.44875001907348633, "reward_judge_quality_std": 0.2125651240348816, "reward_total_composite_mean": 0.7084506750106812, "reward_total_composite_std": 0.12094394117593765} {"timestamp_utc": "2026-04-12T23:03:39Z", "mode": "train", "global_step": 347, "epoch": 0.03485685585133099, "loss": -0.018, "grad_norm": 4.9571099281311035, "learning_rate": 8.951515151515153e-06, "num_tokens": 671556.0, "completions/mean_length": 229.375, "completions/min_length": 207.0, "completions/max_length": 253.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 229.375, "completions/min_terminated_length": 207.0, "completions/max_terminated_length": 253.0, "rewards/meter/mean": 0.8600941896438599, "rewards/meter/std": 0.20765133202075958, "rewards/count_adherence/mean": 0.8541666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.983379602432251, "rewards/repeat_soft/std": 0.019175810739398003, "rewards/judge_quality/mean": 0.2549999952316284, "rewards/judge_quality/std": 0.111867256462574, "rewards/total_composite/mean": 0.6900053024291992, "rewards/total_composite/std": 0.11207691580057144, "reward": 0.6900053024291992, "reward_std": 0.11207691580057144, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15204204618930817, "sampling/sampling_logp_difference/max": 1.2423806190490723, "sampling/importance_sampling_ratio/min": 0.28869614005088806, "sampling/importance_sampling_ratio/mean": 1.0324969291687012, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9544739127159119, "clip_ratio/low_mean": 0.04563032742589712, "clip_ratio/low_min": 0.04563032742589712, "clip_ratio/high_mean": 0.08890281897038221, "clip_ratio/high_max": 0.08890281897038221, "clip_ratio/region_mean": 0.13453314639627934, "reward_total_mean": 0.6900053024291992, "reward_meter_mean": 0.8600941896438599, "reward_meter_std": 0.20765133202075958, "reward_count_adherence_mean": 0.8541666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.983379602432251, "reward_repeat_soft_std": 0.019175810739398003, "reward_judge_quality_mean": 0.2549999952316284, "reward_judge_quality_std": 0.111867256462574, "reward_total_composite_mean": 0.6900053024291992, "reward_total_composite_std": 0.11207691580057144} {"timestamp_utc": "2026-04-12T23:03:46Z", "mode": "train", "global_step": 348, "epoch": 0.03495730788548468, "loss": 0.0728, "grad_norm": 12.619386672973633, "learning_rate": 8.94848484848485e-06, "num_tokens": 673114.0, "completions/mean_length": 51.75, "completions/min_length": 43.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.75, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.5398725867271423, "rewards/meter/std": 0.312899112701416, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9983410239219666, "rewards/repeat_soft/std": 0.0023672564420849085, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.534404993057251, "rewards/total_composite/std": 0.24965423345565796, "reward": 0.534404993057251, "reward_std": 0.24965423345565796, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18539148569107056, "sampling/sampling_logp_difference/max": 1.451909065246582, "sampling/importance_sampling_ratio/min": 0.23412291705608368, "sampling/importance_sampling_ratio/mean": 1.0497229099273682, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.462804928421974, "clip_ratio/low_mean": 0.046489336527884007, "clip_ratio/low_min": 0.046489336527884007, "clip_ratio/high_mean": 0.11662086471915245, "clip_ratio/high_max": 0.11662086471915245, "clip_ratio/region_mean": 0.16311020124703646, "reward_total_mean": 0.534404993057251, "reward_meter_mean": 0.5398725867271423, "reward_meter_std": 0.312899112701416, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9983410239219666, "reward_repeat_soft_std": 0.0023672564420849085, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.534404993057251, "reward_total_composite_std": 0.24965423345565796} {"timestamp_utc": "2026-04-12T23:03:52Z", "mode": "train", "global_step": 349, "epoch": 0.03505775991963837, "loss": 0.0622, "grad_norm": 10.906970024108887, "learning_rate": 8.945454545454546e-06, "num_tokens": 674879.0, "completions/mean_length": 57.625, "completions/min_length": 43.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.625, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9893198013305664, "rewards/meter/std": 0.004541551228612661, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9823523759841919, "rewards/repeat_soft/std": 0.020514938980340958, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.8756791353225708, "rewards/total_composite/std": 0.07663074135780334, "reward": 0.8756791353225708, "reward_std": 0.07663072645664215, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18858855962753296, "sampling/sampling_logp_difference/max": 2.2457327842712402, "sampling/importance_sampling_ratio/min": 0.10584994405508041, "sampling/importance_sampling_ratio/mean": 1.0708712339401245, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3365357518196106, "clip_ratio/low_mean": 0.07978622894734144, "clip_ratio/low_min": 0.07978622894734144, "clip_ratio/high_mean": 0.06785903312265873, "clip_ratio/high_max": 0.06785903312265873, "clip_ratio/region_mean": 0.14764526207000017, "reward_total_mean": 0.8756791353225708, "reward_meter_mean": 0.9893198013305664, "reward_meter_std": 0.004541551228612661, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9823523759841919, "reward_repeat_soft_std": 0.020514938980340958, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.8756791353225708, "reward_total_composite_std": 0.07663074135780334} {"timestamp_utc": "2026-04-12T23:03:59Z", "mode": "train", "global_step": 350, "epoch": 0.03515821195379206, "loss": -0.1022, "grad_norm": 13.024283409118652, "learning_rate": 8.942424242424243e-06, "num_tokens": 676586.0, "completions/mean_length": 45.375, "completions/min_length": 36.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.375, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.8041423559188843, "rewards/meter/std": 0.1913151890039444, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9895690083503723, "rewards/repeat_soft/std": 0.013956138864159584, "rewards/judge_quality/mean": 0.5087499618530273, "rewards/judge_quality/std": 0.16617010533809662, "rewards/total_composite/mean": 0.7634459733963013, "rewards/total_composite/std": 0.11481312662363052, "reward": 0.7634459733963013, "reward_std": 0.11481311917304993, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1655021607875824, "sampling/sampling_logp_difference/max": 1.4025945663452148, "sampling/importance_sampling_ratio/min": 0.24595798552036285, "sampling/importance_sampling_ratio/mean": 1.0174808502197266, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.570731744170189, "clip_ratio/low_mean": 0.07530837319791317, "clip_ratio/low_min": 0.07530837319791317, "clip_ratio/high_mean": 0.07534897048026323, "clip_ratio/high_max": 0.07534897048026323, "clip_ratio/region_mean": 0.1506573436781764, "reward_total_mean": 0.7634459733963013, "reward_meter_mean": 0.8041423559188843, "reward_meter_std": 0.1913151890039444, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9895690083503723, "reward_repeat_soft_std": 0.013956138864159584, "reward_judge_quality_mean": 0.5087499618530273, "reward_judge_quality_std": 0.16617010533809662, "reward_total_composite_mean": 0.7634459733963013, "reward_total_composite_std": 0.11481312662363052} {"timestamp_utc": "2026-04-12T23:04:58Z", "mode": "eval", "global_step": 350, "epoch": 0.03515821195379206, "eval_loss": NaN, "eval_runtime": 58.9512, "eval_samples_per_second": 1.357, "eval_steps_per_second": 0.17, "eval_num_tokens": 676586.0, "eval_completions/mean_length": 98.5625, "eval_completions/min_length": 30.1, "eval_completions/max_length": 256.6, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 87.48214340209961, "eval_completions/min_terminated_length": 30.1, "eval_completions/max_terminated_length": 174.1, "eval_rewards/meter/mean": 0.6642295598983765, "eval_rewards/meter/std": 0.3766138806939125, "eval_rewards/count_adherence/mean": 0.9520833373069764, "eval_rewards/count_adherence/std": 0.10743088461458683, "eval_rewards/hard_gate/mean": 0.9, "eval_rewards/hard_gate/std": 0.20411194264888763, "eval_rewards/repeat_soft/mean": 0.9861673235893249, "eval_rewards/repeat_soft/std": 0.01624320219270885, "eval_rewards/judge_quality/mean": 0.4074999988079071, "eval_rewards/judge_quality/std": 0.19590667337179185, "eval_rewards/total_composite/mean": 0.6107910811901093, "eval_rewards/total_composite/std": 0.25582760721445086, "eval_reward": 0.6107910811901093, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.13716577216982842, "eval_sampling/sampling_logp_difference/max": 1.1873673439025878, "eval_sampling/importance_sampling_ratio/min": 0.3128700226545334, "eval_sampling/importance_sampling_ratio/mean": 1.037182080745697, "eval_sampling/importance_sampling_ratio/max": 1.6084635376930236, "eval_entropy": 2.105841875076294, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6107910811901093, "eval_reward_meter_mean": 0.6642295598983765, "eval_reward_meter_std": 0.3766138806939125, "eval_reward_count_adherence_mean": 0.9520833373069764, "eval_reward_count_adherence_std": 0.10743088461458683, "eval_reward_hard_gate_mean": 0.9, "eval_reward_hard_gate_std": 0.20411194264888763, "eval_reward_repeat_soft_mean": 0.9861673235893249, "eval_reward_repeat_soft_std": 0.01624320219270885, "eval_reward_judge_quality_mean": 0.4074999988079071, "eval_reward_judge_quality_std": 0.19590667337179185, "eval_reward_total_composite_mean": 0.6107910811901093, "eval_reward_total_composite_std": 0.25582760721445086} {"timestamp_utc": "2026-04-12T23:05:06Z", "mode": "train", "global_step": 351, "epoch": 0.035258663987945756, "loss": 0.0726, "grad_norm": 17.249513626098633, "learning_rate": 8.93939393939394e-06, "num_tokens": 678025.0, "completions/mean_length": 29.875, "completions/min_length": 23.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.875, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.7305623292922974, "rewards/meter/std": 0.43501585721969604, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.7437499761581421, "rewards/judge_quality/std": 0.2432481348514557, "rewards/total_composite/mean": 0.7981280088424683, "rewards/total_composite/std": 0.21738097071647644, "reward": 0.7981280088424683, "reward_std": 0.21738095581531525, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16106292605400085, "sampling/sampling_logp_difference/max": 0.9539041519165039, "sampling/importance_sampling_ratio/min": 0.3852340877056122, "sampling/importance_sampling_ratio/mean": 1.03627347946167, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.773281142115593, "clip_ratio/low_mean": 0.008417508564889431, "clip_ratio/low_min": 0.008417508564889431, "clip_ratio/high_mean": 0.12614015396684408, "clip_ratio/high_max": 0.12614015396684408, "clip_ratio/region_mean": 0.1345576625317335, "reward_total_mean": 0.7981280088424683, "reward_meter_mean": 0.7305623292922974, "reward_meter_std": 0.43501585721969604, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.7437499761581421, "reward_judge_quality_std": 0.2432481348514557, "reward_total_composite_mean": 0.7981280088424683, "reward_total_composite_std": 0.21738097071647644} {"timestamp_utc": "2026-04-12T23:05:13Z", "mode": "train", "global_step": 352, "epoch": 0.03535911602209945, "loss": 0.0333, "grad_norm": 14.175741195678711, "learning_rate": 8.936363636363638e-06, "num_tokens": 679623.0, "completions/mean_length": 46.75, "completions/min_length": 29.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.75, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.904102623462677, "rewards/meter/std": 0.20507816970348358, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9842955470085144, "rewards/repeat_soft/std": 0.037771038711071014, "rewards/judge_quality/mean": 0.690000057220459, "rewards/judge_quality/std": 0.281475692987442, "rewards/total_composite/mean": 0.8622757196426392, "rewards/total_composite/std": 0.12282069772481918, "reward": 0.8622757196426392, "reward_std": 0.12282069027423859, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1974189728498459, "sampling/sampling_logp_difference/max": 2.717233657836914, "sampling/importance_sampling_ratio/min": 0.0660572424530983, "sampling/importance_sampling_ratio/mean": 1.0375105142593384, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.785032495856285, "clip_ratio/low_mean": 0.08132252376526594, "clip_ratio/low_min": 0.08132252376526594, "clip_ratio/high_mean": 0.08908769767731428, "clip_ratio/high_max": 0.08908769767731428, "clip_ratio/region_mean": 0.17041022144258022, "reward_total_mean": 0.8622757196426392, "reward_meter_mean": 0.904102623462677, "reward_meter_std": 0.20507816970348358, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9842955470085144, "reward_repeat_soft_std": 0.037771038711071014, "reward_judge_quality_mean": 0.690000057220459, "reward_judge_quality_std": 0.281475692987442, "reward_total_composite_mean": 0.8622757196426392, "reward_total_composite_std": 0.12282069772481918} {"timestamp_utc": "2026-04-12T23:05:19Z", "mode": "train", "global_step": 353, "epoch": 0.03545956805625314, "loss": 0.0363, "grad_norm": 11.490890502929688, "learning_rate": 8.933333333333333e-06, "num_tokens": 681511.0, "completions/mean_length": 57.0, "completions/min_length": 42.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.0, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.6460087299346924, "rewards/meter/std": 0.30132240056991577, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9941244721412659, "rewards/repeat_soft/std": 0.009034770540893078, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.7433663606643677, "rewards/total_composite/std": 0.15035519003868103, "reward": 0.7433663606643677, "reward_std": 0.15035520493984222, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17570988833904266, "sampling/sampling_logp_difference/max": 3.096266508102417, "sampling/importance_sampling_ratio/min": 0.04521770775318146, "sampling/importance_sampling_ratio/mean": 1.0093456506729126, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.434578999876976, "clip_ratio/low_mean": 0.07345836982131004, "clip_ratio/low_min": 0.07345836982131004, "clip_ratio/high_mean": 0.09936208557337523, "clip_ratio/high_max": 0.09936208557337523, "clip_ratio/region_mean": 0.17282045539468527, "reward_total_mean": 0.7433663606643677, "reward_meter_mean": 0.6460087299346924, "reward_meter_std": 0.30132240056991577, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9941244721412659, "reward_repeat_soft_std": 0.009034770540893078, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.7433663606643677, "reward_total_composite_std": 0.15035519003868103} {"timestamp_utc": "2026-04-12T23:05:27Z", "mode": "train", "global_step": 354, "epoch": 0.03556002009040683, "loss": -0.0034, "grad_norm": 20.28631591796875, "learning_rate": 8.930303030303032e-06, "num_tokens": 683119.0, "completions/mean_length": 37.0, "completions/min_length": 30.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.0, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.4639221727848053, "rewards/meter/std": 0.43121805787086487, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9962563514709473, "rewards/repeat_soft/std": 0.00458854204043746, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.6968905925750732, "rewards/total_composite/std": 0.20049096643924713, "reward": 0.6968905925750732, "reward_std": 0.20049096643924713, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17409412562847137, "sampling/sampling_logp_difference/max": 1.5366554260253906, "sampling/importance_sampling_ratio/min": 0.21509931981563568, "sampling/importance_sampling_ratio/mean": 1.0221203565597534, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5846166536211967, "clip_ratio/low_mean": 0.1030505308881402, "clip_ratio/low_min": 0.1030505308881402, "clip_ratio/high_mean": 0.06598432175815105, "clip_ratio/high_max": 0.06598432175815105, "clip_ratio/region_mean": 0.16903485264629126, "reward_total_mean": 0.6968905925750732, "reward_meter_mean": 0.4639221727848053, "reward_meter_std": 0.43121805787086487, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9962563514709473, "reward_repeat_soft_std": 0.00458854204043746, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.6968905925750732, "reward_total_composite_std": 0.20049096643924713} {"timestamp_utc": "2026-04-12T23:05:33Z", "mode": "train", "global_step": 355, "epoch": 0.03566047212456052, "loss": -0.0199, "grad_norm": 12.868624687194824, "learning_rate": 8.927272727272728e-06, "num_tokens": 684631.0, "completions/mean_length": 47.0, "completions/min_length": 33.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.5254199504852295, "rewards/meter/std": 0.2510526478290558, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9956212043762207, "rewards/repeat_soft/std": 0.011898322962224483, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.6018760800361633, "rewards/total_composite/std": 0.13567426800727844, "reward": 0.6018760800361633, "reward_std": 0.13567426800727844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2015838921070099, "sampling/sampling_logp_difference/max": 1.4846014976501465, "sampling/importance_sampling_ratio/min": 0.22659263014793396, "sampling/importance_sampling_ratio/mean": 1.0348131656646729, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3785996586084366, "clip_ratio/low_mean": 0.06928872503340244, "clip_ratio/low_min": 0.06928872503340244, "clip_ratio/high_mean": 0.11202024668455124, "clip_ratio/high_max": 0.11202024668455124, "clip_ratio/region_mean": 0.18130897171795368, "reward_total_mean": 0.6018760800361633, "reward_meter_mean": 0.5254199504852295, "reward_meter_std": 0.2510526478290558, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9956212043762207, "reward_repeat_soft_std": 0.011898322962224483, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.6018760800361633, "reward_total_composite_std": 0.13567426800727844} {"timestamp_utc": "2026-04-12T23:05:43Z", "mode": "train", "global_step": 356, "epoch": 0.035760924158714215, "loss": -0.0152, "grad_norm": 6.300727367401123, "learning_rate": 8.924242424242425e-06, "num_tokens": 687037.0, "completions/mean_length": 142.75, "completions/min_length": 88.0, "completions/max_length": 176.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.75, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 176.0, "rewards/meter/mean": 0.7624008059501648, "rewards/meter/std": 0.3128993809223175, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.988207221031189, "rewards/repeat_soft/std": 0.01710481382906437, "rewards/judge_quality/mean": 0.24250000715255737, "rewards/judge_quality/std": 0.11792854964733124, "rewards/total_composite/mean": 0.641213595867157, "rewards/total_composite/std": 0.14200136065483093, "reward": 0.641213595867157, "reward_std": 0.14200134575366974, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19974730908870697, "sampling/sampling_logp_difference/max": 1.3332531452178955, "sampling/importance_sampling_ratio/min": 0.2636182904243469, "sampling/importance_sampling_ratio/mean": 1.0397799015045166, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.080691620707512, "clip_ratio/low_mean": 0.05674755573272705, "clip_ratio/low_min": 0.05674755573272705, "clip_ratio/high_mean": 0.08736725524067879, "clip_ratio/high_max": 0.08736725524067879, "clip_ratio/region_mean": 0.14411481097340584, "reward_total_mean": 0.641213595867157, "reward_meter_mean": 0.7624008059501648, "reward_meter_std": 0.3128993809223175, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.988207221031189, "reward_repeat_soft_std": 0.01710481382906437, "reward_judge_quality_mean": 0.24250000715255737, "reward_judge_quality_std": 0.11792854964733124, "reward_total_composite_mean": 0.641213595867157, "reward_total_composite_std": 0.14200136065483093} {"timestamp_utc": "2026-04-12T23:05:51Z", "mode": "train", "global_step": 357, "epoch": 0.0358613761928679, "loss": -0.0309, "grad_norm": 4.9387359619140625, "learning_rate": 8.921212121212122e-06, "num_tokens": 689909.0, "completions/mean_length": 178.0, "completions/min_length": 110.0, "completions/max_length": 214.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 178.0, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 214.0, "rewards/meter/mean": 0.8042426705360413, "rewards/meter/std": 0.27287283539772034, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9646809101104736, "rewards/repeat_soft/std": 0.031365200877189636, "rewards/judge_quality/mean": 0.20000000298023224, "rewards/judge_quality/std": 0.05345224589109421, "rewards/total_composite/mean": 0.642127275466919, "rewards/total_composite/std": 0.10756617039442062, "reward": 0.642127275466919, "reward_std": 0.10756617784500122, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17486810684204102, "sampling/sampling_logp_difference/max": 1.5930166244506836, "sampling/importance_sampling_ratio/min": 0.20331138372421265, "sampling/importance_sampling_ratio/mean": 1.033842921257019, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6393115520477295, "clip_ratio/low_mean": 0.049981351010501385, "clip_ratio/low_min": 0.049981351010501385, "clip_ratio/high_mean": 0.07008560374379158, "clip_ratio/high_max": 0.07008560374379158, "clip_ratio/region_mean": 0.12006695475429296, "reward_total_mean": 0.642127275466919, "reward_meter_mean": 0.8042426705360413, "reward_meter_std": 0.27287283539772034, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9646809101104736, "reward_repeat_soft_std": 0.031365200877189636, "reward_judge_quality_mean": 0.20000000298023224, "reward_judge_quality_std": 0.05345224589109421, "reward_total_composite_mean": 0.642127275466919, "reward_total_composite_std": 0.10756617039442062} {"timestamp_utc": "2026-04-12T23:06:04Z", "mode": "train", "global_step": 358, "epoch": 0.0359618282270216, "loss": -0.1389, "grad_norm": 3.6793134212493896, "learning_rate": 8.91818181818182e-06, "num_tokens": 691651.0, "completions/mean_length": 123.75, "completions/min_length": 47.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 68.28572082519531, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.38434866070747375, "rewards/meter/std": 0.3749138116836548, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.24800792336463928, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9927225112915039, "rewards/repeat_soft/std": 0.005280505865812302, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.13274572789669037, "rewards/total_composite/mean": 0.44656285643577576, "rewards/total_composite/std": 0.2353944182395935, "reward": 0.44656285643577576, "reward_std": 0.2353944182395935, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1684454083442688, "sampling/sampling_logp_difference/max": 1.6209821701049805, "sampling/importance_sampling_ratio/min": 0.19770441949367523, "sampling/importance_sampling_ratio/mean": 1.0533591508865356, "sampling/importance_sampling_ratio/max": 1.8893535137176514, "entropy": 1.4987786561250687, "clip_ratio/low_mean": 0.04511632025241852, "clip_ratio/low_min": 0.04511632025241852, "clip_ratio/high_mean": 0.0687566613778472, "clip_ratio/high_max": 0.0687566613778472, "clip_ratio/region_mean": 0.11387298163026571, "reward_total_mean": 0.44656285643577576, "reward_meter_mean": 0.38434866070747375, "reward_meter_std": 0.3749138116836548, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.24800792336463928, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9927225112915039, "reward_repeat_soft_std": 0.005280505865812302, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.13274572789669037, "reward_total_composite_mean": 0.44656285643577576, "reward_total_composite_std": 0.2353944182395935} {"timestamp_utc": "2026-04-12T23:06:17Z", "mode": "train", "global_step": 359, "epoch": 0.03606228026117529, "loss": -0.2286, "grad_norm": 2.2040886878967285, "learning_rate": 8.915151515151515e-06, "num_tokens": 693995.0, "completions/mean_length": 174.0, "completions/min_length": 98.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 125.71429443359375, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.85038161277771, "rewards/meter/std": 0.14151319861412048, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9739665985107422, "rewards/repeat_soft/std": 0.024584896862506866, "rewards/judge_quality/mean": 0.23624999821186066, "rewards/judge_quality/std": 0.1356400102376938, "rewards/total_composite/mean": 0.6161206960678101, "rewards/total_composite/std": 0.26188230514526367, "reward": 0.6161206960678101, "reward_std": 0.26188230514526367, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19410815834999084, "sampling/sampling_logp_difference/max": 1.6937966346740723, "sampling/importance_sampling_ratio/min": 0.18382030725479126, "sampling/importance_sampling_ratio/mean": 1.049407958984375, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4368839263916016, "clip_ratio/low_mean": 0.01354166679084301, "clip_ratio/low_min": 0.01354166679084301, "clip_ratio/high_mean": 0.10565605014562607, "clip_ratio/high_max": 0.10565605014562607, "clip_ratio/region_mean": 0.11919771693646908, "reward_total_mean": 0.6161206960678101, "reward_meter_mean": 0.85038161277771, "reward_meter_std": 0.14151319861412048, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.2651650309562683, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9739665985107422, "reward_repeat_soft_std": 0.024584896862506866, "reward_judge_quality_mean": 0.23624999821186066, "reward_judge_quality_std": 0.1356400102376938, "reward_total_composite_mean": 0.6161206960678101, "reward_total_composite_std": 0.26188230514526367} {"timestamp_utc": "2026-04-12T23:06:24Z", "mode": "train", "global_step": 360, "epoch": 0.03616273229532898, "loss": 0.0856, "grad_norm": 8.71314525604248, "learning_rate": 8.912121212121214e-06, "num_tokens": 696019.0, "completions/mean_length": 89.0, "completions/min_length": 66.0, "completions/max_length": 119.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.4989315867424011, "rewards/meter/std": 0.34187570214271545, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9823164939880371, "rewards/repeat_soft/std": 0.017562491819262505, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.12631450593471527, "rewards/total_composite/mean": 0.47058671712875366, "rewards/total_composite/std": 0.31852537393569946, "reward": 0.47058671712875366, "reward_std": 0.31852537393569946, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1735786646604538, "sampling/sampling_logp_difference/max": 1.6762304306030273, "sampling/importance_sampling_ratio/min": 0.1870778352022171, "sampling/importance_sampling_ratio/mean": 1.0335789918899536, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2738180607557297, "clip_ratio/low_mean": 0.04161112755537033, "clip_ratio/low_min": 0.04161112755537033, "clip_ratio/high_mean": 0.10257037915289402, "clip_ratio/high_max": 0.10257037915289402, "clip_ratio/region_mean": 0.14418150670826435, "reward_total_mean": 0.47058671712875366, "reward_meter_mean": 0.4989315867424011, "reward_meter_std": 0.34187570214271545, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9823164939880371, "reward_repeat_soft_std": 0.017562491819262505, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.12631450593471527, "reward_total_composite_mean": 0.47058671712875366, "reward_total_composite_std": 0.31852537393569946} {"timestamp_utc": "2026-04-12T23:06:32Z", "mode": "train", "global_step": 361, "epoch": 0.036263184329482674, "loss": 0.0837, "grad_norm": 8.904894828796387, "learning_rate": 8.90909090909091e-06, "num_tokens": 698070.0, "completions/mean_length": 93.375, "completions/min_length": 83.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.375, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.9904606342315674, "rewards/meter/std": 0.00592720415443182, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9883317351341248, "rewards/repeat_soft/std": 0.010305705480277538, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.7976654767990112, "rewards/total_composite/std": 0.03205154463648796, "reward": 0.7976654767990112, "reward_std": 0.032051533460617065, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18020442128181458, "sampling/sampling_logp_difference/max": 2.1122875213623047, "sampling/importance_sampling_ratio/min": 0.12096095085144043, "sampling/importance_sampling_ratio/mean": 1.037337303161621, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.339341402053833, "clip_ratio/low_mean": 0.04732718039304018, "clip_ratio/low_min": 0.04732718039304018, "clip_ratio/high_mean": 0.09408726077526808, "clip_ratio/high_max": 0.09408726077526808, "clip_ratio/region_mean": 0.14141444116830826, "reward_total_mean": 0.7976654767990112, "reward_meter_mean": 0.9904606342315674, "reward_meter_std": 0.00592720415443182, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9883317351341248, "reward_repeat_soft_std": 0.010305705480277538, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.7976654767990112, "reward_total_composite_std": 0.03205154463648796} {"timestamp_utc": "2026-04-12T23:06:38Z", "mode": "train", "global_step": 362, "epoch": 0.03636363636363636, "loss": -0.0171, "grad_norm": 21.15106964111328, "learning_rate": 8.906060606060607e-06, "num_tokens": 699580.0, "completions/mean_length": 27.75, "completions/min_length": 22.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.75, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.33231261372566223, "rewards/meter/std": 0.4086683392524719, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9713224172592163, "rewards/repeat_soft/std": 0.034012529999017715, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.5425479412078857, "rewards/total_composite/std": 0.22006665170192719, "reward": 0.5425479412078857, "reward_std": 0.22006665170192719, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21327532827854156, "sampling/sampling_logp_difference/max": 2.998166084289551, "sampling/importance_sampling_ratio/min": 0.04987845942378044, "sampling/importance_sampling_ratio/mean": 0.9876313805580139, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2308171913027763, "clip_ratio/low_mean": 0.10922979842871428, "clip_ratio/low_min": 0.10922979842871428, "clip_ratio/high_mean": 0.0754944235086441, "clip_ratio/high_max": 0.0754944235086441, "clip_ratio/region_mean": 0.18472422193735838, "reward_total_mean": 0.5425479412078857, "reward_meter_mean": 0.33231261372566223, "reward_meter_std": 0.4086683392524719, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9713224172592163, "reward_repeat_soft_std": 0.034012529999017715, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.5425479412078857, "reward_total_composite_std": 0.22006665170192719} {"timestamp_utc": "2026-04-12T23:06:47Z", "mode": "train", "global_step": 363, "epoch": 0.036464088397790057, "loss": -0.0207, "grad_norm": 8.160727500915527, "learning_rate": 8.903030303030304e-06, "num_tokens": 701935.0, "completions/mean_length": 112.375, "completions/min_length": 73.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.375, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.4048033356666565, "rewards/meter/std": 0.346056193113327, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.18600596487522125, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9689017534255981, "rewards/repeat_soft/std": 0.031244073063135147, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.20119288563728333, "rewards/total_composite/mean": 0.5642391443252563, "rewards/total_composite/std": 0.21127977967262268, "reward": 0.5642391443252563, "reward_std": 0.21127977967262268, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17640168964862823, "sampling/sampling_logp_difference/max": 2.063220739364624, "sampling/importance_sampling_ratio/min": 0.12704412639141083, "sampling/importance_sampling_ratio/mean": 1.0345640182495117, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9750403612852097, "clip_ratio/low_mean": 0.09460058622062206, "clip_ratio/low_min": 0.09460058622062206, "clip_ratio/high_mean": 0.07094144262373447, "clip_ratio/high_max": 0.07094144262373447, "clip_ratio/region_mean": 0.16554202884435654, "reward_total_mean": 0.5642391443252563, "reward_meter_mean": 0.4048033356666565, "reward_meter_std": 0.346056193113327, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.18600596487522125, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9689017534255981, "reward_repeat_soft_std": 0.031244073063135147, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.20119288563728333, "reward_total_composite_mean": 0.5642391443252563, "reward_total_composite_std": 0.21127977967262268} {"timestamp_utc": "2026-04-12T23:06:54Z", "mode": "train", "global_step": 364, "epoch": 0.036564540431943744, "loss": 0.015, "grad_norm": 13.294562339782715, "learning_rate": 8.900000000000001e-06, "num_tokens": 703481.0, "completions/mean_length": 47.25, "completions/min_length": 38.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.25, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.861323356628418, "rewards/meter/std": 0.25819140672683716, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9726259708404541, "rewards/repeat_soft/std": 0.035471320152282715, "rewards/judge_quality/mean": 0.6274999976158142, "rewards/judge_quality/std": 0.21952873468399048, "rewards/total_composite/mean": 0.8231080770492554, "rewards/total_composite/std": 0.12236202508211136, "reward": 0.8231080770492554, "reward_std": 0.12236204743385315, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14462779462337494, "sampling/sampling_logp_difference/max": 1.2618517875671387, "sampling/importance_sampling_ratio/min": 0.2831292450428009, "sampling/importance_sampling_ratio/mean": 1.0095114707946777, "sampling/importance_sampling_ratio/max": 1.9468475580215454, "entropy": 1.2733398899435997, "clip_ratio/low_mean": 0.05723207117989659, "clip_ratio/low_min": 0.05723207117989659, "clip_ratio/high_mean": 0.0542620662599802, "clip_ratio/high_max": 0.0542620662599802, "clip_ratio/region_mean": 0.1114941374398768, "reward_total_mean": 0.8231080770492554, "reward_meter_mean": 0.861323356628418, "reward_meter_std": 0.25819140672683716, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9726259708404541, "reward_repeat_soft_std": 0.035471320152282715, "reward_judge_quality_mean": 0.6274999976158142, "reward_judge_quality_std": 0.21952873468399048, "reward_total_composite_mean": 0.8231080770492554, "reward_total_composite_std": 0.12236202508211136} {"timestamp_utc": "2026-04-12T23:07:02Z", "mode": "train", "global_step": 365, "epoch": 0.03666499246609744, "loss": 0.0102, "grad_norm": 10.343697547912598, "learning_rate": 8.896969696969697e-06, "num_tokens": 705702.0, "completions/mean_length": 101.625, "completions/min_length": 66.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.625, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.7602600455284119, "rewards/meter/std": 0.39141684770584106, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9715206623077393, "rewards/repeat_soft/std": 0.04792093485593796, "rewards/judge_quality/mean": 0.3687500059604645, "rewards/judge_quality/std": 0.186581090092659, "rewards/total_composite/mean": 0.5949800610542297, "rewards/total_composite/std": 0.30850595235824585, "reward": 0.5949800610542297, "reward_std": 0.30850595235824585, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16934095323085785, "sampling/sampling_logp_difference/max": 2.024158477783203, "sampling/importance_sampling_ratio/min": 0.13210496306419373, "sampling/importance_sampling_ratio/mean": 1.0373294353485107, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9023277461528778, "clip_ratio/low_mean": 0.05822381656616926, "clip_ratio/low_min": 0.05822381656616926, "clip_ratio/high_mean": 0.1175051648169756, "clip_ratio/high_max": 0.1175051648169756, "clip_ratio/region_mean": 0.17572898138314486, "reward_total_mean": 0.5949800610542297, "reward_meter_mean": 0.7602600455284119, "reward_meter_std": 0.39141684770584106, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9715206623077393, "reward_repeat_soft_std": 0.04792093485593796, "reward_judge_quality_mean": 0.3687500059604645, "reward_judge_quality_std": 0.186581090092659, "reward_total_composite_mean": 0.5949800610542297, "reward_total_composite_std": 0.30850595235824585} {"timestamp_utc": "2026-04-12T23:07:08Z", "mode": "train", "global_step": 366, "epoch": 0.036765444500251133, "loss": 0.0366, "grad_norm": 13.168122291564941, "learning_rate": 8.893939393939394e-06, "num_tokens": 707405.0, "completions/mean_length": 52.875, "completions/min_length": 31.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.875, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.7133063673973083, "rewards/meter/std": 0.41369760036468506, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9842491745948792, "rewards/repeat_soft/std": 0.01343735121190548, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.6789127588272095, "rewards/total_composite/std": 0.21660472452640533, "reward": 0.6789127588272095, "reward_std": 0.21660472452640533, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19049054384231567, "sampling/sampling_logp_difference/max": 1.1834940910339355, "sampling/importance_sampling_ratio/min": 0.32206180691719055, "sampling/importance_sampling_ratio/mean": 1.0316791534423828, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0714560449123383, "clip_ratio/low_mean": 0.03320567216724157, "clip_ratio/low_min": 0.03320567216724157, "clip_ratio/high_mean": 0.14752879552543163, "clip_ratio/high_max": 0.14752879552543163, "clip_ratio/region_mean": 0.1807344676926732, "reward_total_mean": 0.6789127588272095, "reward_meter_mean": 0.7133063673973083, "reward_meter_std": 0.41369760036468506, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9842491745948792, "reward_repeat_soft_std": 0.01343735121190548, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.6789127588272095, "reward_total_composite_std": 0.21660472452640533} {"timestamp_utc": "2026-04-12T23:07:21Z", "mode": "train", "global_step": 367, "epoch": 0.03686589653440482, "loss": -0.2352, "grad_norm": 2.335317373275757, "learning_rate": 8.890909090909091e-06, "num_tokens": 709661.0, "completions/mean_length": 296.0, "completions/min_length": 131.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 166.40000915527344, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 204.0, "rewards/meter/mean": 0.3070393204689026, "rewards/meter/std": 0.24342399835586548, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.1259881556034088, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9829114079475403, "rewards/repeat_soft/std": 0.025337008759379387, "rewards/judge_quality/mean": 0.17500001192092896, "rewards/judge_quality/std": 0.08864052593708038, "rewards/total_composite/mean": 0.35196229815483093, "rewards/total_composite/std": 0.2381960153579712, "reward": 0.35196229815483093, "reward_std": 0.23819600045681, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21107670664787292, "sampling/sampling_logp_difference/max": 1.1658973693847656, "sampling/importance_sampling_ratio/min": 0.3116428852081299, "sampling/importance_sampling_ratio/mean": 1.0484986305236816, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9273861348628998, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.12177976779639721, "clip_ratio/high_max": 0.12177976779639721, "clip_ratio/region_mean": 0.12177976779639721, "reward_total_mean": 0.35196229815483093, "reward_meter_mean": 0.3070393204689026, "reward_meter_std": 0.24342399835586548, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.1259881556034088, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9829114079475403, "reward_repeat_soft_std": 0.025337008759379387, "reward_judge_quality_mean": 0.17500001192092896, "reward_judge_quality_std": 0.08864052593708038, "reward_total_composite_mean": 0.35196229815483093, "reward_total_composite_std": 0.2381960153579712} {"timestamp_utc": "2026-04-12T23:07:27Z", "mode": "train", "global_step": 368, "epoch": 0.036966348568558516, "loss": 0.0316, "grad_norm": 13.078278541564941, "learning_rate": 8.887878787878789e-06, "num_tokens": 711380.0, "completions/mean_length": 49.875, "completions/min_length": 35.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.875, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.43339619040489197, "rewards/meter/std": 0.3887919783592224, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9824534058570862, "rewards/repeat_soft/std": 0.01932821236550808, "rewards/judge_quality/mean": 0.518750011920929, "rewards/judge_quality/std": 0.26637449860572815, "rewards/total_composite/mean": 0.5988986492156982, "rewards/total_composite/std": 0.16298170387744904, "reward": 0.5988986492156982, "reward_std": 0.16298170387744904, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17455117404460907, "sampling/sampling_logp_difference/max": 1.530686378479004, "sampling/importance_sampling_ratio/min": 0.2163870930671692, "sampling/importance_sampling_ratio/mean": 1.0397918224334717, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.003060296177864, "clip_ratio/low_mean": 0.07367203710600734, "clip_ratio/low_min": 0.07367203710600734, "clip_ratio/high_mean": 0.07486280612647533, "clip_ratio/high_max": 0.07486280612647533, "clip_ratio/region_mean": 0.14853484323248267, "reward_total_mean": 0.5988986492156982, "reward_meter_mean": 0.43339619040489197, "reward_meter_std": 0.3887919783592224, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9824534058570862, "reward_repeat_soft_std": 0.01932821236550808, "reward_judge_quality_mean": 0.518750011920929, "reward_judge_quality_std": 0.26637449860572815, "reward_total_composite_mean": 0.5988986492156982, "reward_total_composite_std": 0.16298170387744904} {"timestamp_utc": "2026-04-12T23:07:35Z", "mode": "train", "global_step": 369, "epoch": 0.037066800602712204, "loss": -0.0268, "grad_norm": 7.859500885009766, "learning_rate": 8.884848484848486e-06, "num_tokens": 713325.0, "completions/mean_length": 86.125, "completions/min_length": 59.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.125, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.581904411315918, "rewards/meter/std": 0.35868942737579346, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9798800945281982, "rewards/repeat_soft/std": 0.017094559967517853, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.606594979763031, "rewards/total_composite/std": 0.15068362653255463, "reward": 0.606594979763031, "reward_std": 0.15068362653255463, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17260082066059113, "sampling/sampling_logp_difference/max": 1.2875146865844727, "sampling/importance_sampling_ratio/min": 0.27595576643943787, "sampling/importance_sampling_ratio/mean": 1.040467381477356, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3359992802143097, "clip_ratio/low_mean": 0.0797077352181077, "clip_ratio/low_min": 0.0797077352181077, "clip_ratio/high_mean": 0.05408669263124466, "clip_ratio/high_max": 0.05408669263124466, "clip_ratio/region_mean": 0.13379442784935236, "reward_total_mean": 0.606594979763031, "reward_meter_mean": 0.581904411315918, "reward_meter_std": 0.35868942737579346, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9798800945281982, "reward_repeat_soft_std": 0.017094559967517853, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.606594979763031, "reward_total_composite_std": 0.15068362653255463} {"timestamp_utc": "2026-04-12T23:07:41Z", "mode": "train", "global_step": 370, "epoch": 0.0371672526368659, "loss": 0.1022, "grad_norm": 20.564205169677734, "learning_rate": 8.881818181818183e-06, "num_tokens": 714859.0, "completions/mean_length": 33.75, "completions/min_length": 27.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.75, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.13933514058589935, "rewards/meter/std": 0.09607045352458954, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9930397272109985, "rewards/repeat_soft/std": 0.016566580161452293, "rewards/judge_quality/mean": 0.8562500476837158, "rewards/judge_quality/std": 0.1647454798221588, "rewards/total_composite/mean": 0.5688798427581787, "rewards/total_composite/std": 0.0700153186917305, "reward": 0.5688798427581787, "reward_std": 0.0700153112411499, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1568070501089096, "sampling/sampling_logp_difference/max": 1.3572371006011963, "sampling/importance_sampling_ratio/min": 0.25737088918685913, "sampling/importance_sampling_ratio/mean": 1.0030126571655273, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9065039977431297, "clip_ratio/low_mean": 0.06757478788495064, "clip_ratio/low_min": 0.06757478788495064, "clip_ratio/high_mean": 0.08786436170339584, "clip_ratio/high_max": 0.08786436170339584, "clip_ratio/region_mean": 0.15543914958834648, "reward_total_mean": 0.5688798427581787, "reward_meter_mean": 0.13933514058589935, "reward_meter_std": 0.09607045352458954, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9930397272109985, "reward_repeat_soft_std": 0.016566580161452293, "reward_judge_quality_mean": 0.8562500476837158, "reward_judge_quality_std": 0.1647454798221588, "reward_total_composite_mean": 0.5688798427581787, "reward_total_composite_std": 0.0700153186917305} {"timestamp_utc": "2026-04-12T23:07:48Z", "mode": "train", "global_step": 371, "epoch": 0.037267704671019586, "loss": -0.0577, "grad_norm": 12.491142272949219, "learning_rate": 8.87878787878788e-06, "num_tokens": 716613.0, "completions/mean_length": 57.25, "completions/min_length": 40.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.25, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.859676718711853, "rewards/meter/std": 0.24414286017417908, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9443264007568359, "rewards/repeat_soft/std": 0.058333687484264374, "rewards/judge_quality/mean": 0.36374998092651367, "rewards/judge_quality/std": 0.17935898900032043, "rewards/total_composite/mean": 0.740412175655365, "rewards/total_composite/std": 0.11401320993900299, "reward": 0.740412175655365, "reward_std": 0.11401321738958359, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13325148820877075, "sampling/sampling_logp_difference/max": 1.4858386516571045, "sampling/importance_sampling_ratio/min": 0.22631247341632843, "sampling/importance_sampling_ratio/mean": 1.0287809371948242, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2631319910287857, "clip_ratio/low_mean": 0.05857211444526911, "clip_ratio/low_min": 0.05857211444526911, "clip_ratio/high_mean": 0.04645686410367489, "clip_ratio/high_max": 0.04645686410367489, "clip_ratio/region_mean": 0.105028978548944, "reward_total_mean": 0.740412175655365, "reward_meter_mean": 0.859676718711853, "reward_meter_std": 0.24414286017417908, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9443264007568359, "reward_repeat_soft_std": 0.058333687484264374, "reward_judge_quality_mean": 0.36374998092651367, "reward_judge_quality_std": 0.17935898900032043, "reward_total_composite_mean": 0.740412175655365, "reward_total_composite_std": 0.11401320993900299} {"timestamp_utc": "2026-04-12T23:07:54Z", "mode": "train", "global_step": 372, "epoch": 0.03736815670517328, "loss": 0.0693, "grad_norm": 23.785154342651367, "learning_rate": 8.875757575757576e-06, "num_tokens": 718147.0, "completions/mean_length": 28.75, "completions/min_length": 24.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.75, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.5978069305419922, "rewards/meter/std": 0.3759942650794983, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9332931041717529, "rewards/repeat_soft/std": 0.0332900732755661, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.6330924034118652, "rewards/total_composite/std": 0.15918341279029846, "reward": 0.6330924034118652, "reward_std": 0.15918339788913727, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17136311531066895, "sampling/sampling_logp_difference/max": 1.3128814697265625, "sampling/importance_sampling_ratio/min": 0.2690436840057373, "sampling/importance_sampling_ratio/mean": 1.0302149057388306, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4581508859992027, "clip_ratio/low_mean": 0.04330963268876076, "clip_ratio/low_min": 0.04330963268876076, "clip_ratio/high_mean": 0.09944659005850554, "clip_ratio/high_max": 0.09944659005850554, "clip_ratio/region_mean": 0.1427562227472663, "reward_total_mean": 0.6330924034118652, "reward_meter_mean": 0.5978069305419922, "reward_meter_std": 0.3759942650794983, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9332931041717529, "reward_repeat_soft_std": 0.0332900732755661, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.6330924034118652, "reward_total_composite_std": 0.15918341279029846} {"timestamp_utc": "2026-04-12T23:08:01Z", "mode": "train", "global_step": 373, "epoch": 0.03746860873932697, "loss": 0.0671, "grad_norm": 9.830558776855469, "learning_rate": 8.872727272727275e-06, "num_tokens": 719975.0, "completions/mean_length": 66.5, "completions/min_length": 59.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.48904070258140564, "rewards/meter/std": 0.44804850220680237, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9833070039749146, "rewards/repeat_soft/std": 0.01685311272740364, "rewards/judge_quality/mean": 0.1875, "rewards/judge_quality/std": 0.0517549142241478, "rewards/total_composite/mean": 0.5246490240097046, "rewards/total_composite/std": 0.20595847070217133, "reward": 0.5246490240097046, "reward_std": 0.20595847070217133, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18569175899028778, "sampling/sampling_logp_difference/max": 1.4995861053466797, "sampling/importance_sampling_ratio/min": 0.2232225388288498, "sampling/importance_sampling_ratio/mean": 1.0186481475830078, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1803979575634003, "clip_ratio/low_mean": 0.06942903343588114, "clip_ratio/low_min": 0.06942903343588114, "clip_ratio/high_mean": 0.08886997401714325, "clip_ratio/high_max": 0.08886997401714325, "clip_ratio/region_mean": 0.1582990074530244, "reward_total_mean": 0.5246490240097046, "reward_meter_mean": 0.48904070258140564, "reward_meter_std": 0.44804850220680237, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9833070039749146, "reward_repeat_soft_std": 0.01685311272740364, "reward_judge_quality_mean": 0.1875, "reward_judge_quality_std": 0.0517549142241478, "reward_total_composite_mean": 0.5246490240097046, "reward_total_composite_std": 0.20595847070217133} {"timestamp_utc": "2026-04-12T23:08:08Z", "mode": "train", "global_step": 374, "epoch": 0.03756906077348066, "loss": -0.0364, "grad_norm": 12.802114486694336, "learning_rate": 8.86969696969697e-06, "num_tokens": 721502.0, "completions/mean_length": 42.875, "completions/min_length": 33.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.8647017478942871, "rewards/meter/std": 0.2624070942401886, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9960718154907227, "rewards/repeat_soft/std": 0.0056772087700665, "rewards/judge_quality/mean": 0.32624998688697815, "rewards/judge_quality/std": 0.11350739002227783, "rewards/total_composite/mean": 0.5566011667251587, "rewards/total_composite/std": 0.36000367999076843, "reward": 0.5566011667251587, "reward_std": 0.36000367999076843, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21123836934566498, "sampling/sampling_logp_difference/max": 1.7367315292358398, "sampling/importance_sampling_ratio/min": 0.17609502375125885, "sampling/importance_sampling_ratio/mean": 1.0151444673538208, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.390852525830269, "clip_ratio/low_mean": 0.0661363648250699, "clip_ratio/low_min": 0.0661363648250699, "clip_ratio/high_mean": 0.11709614004939795, "clip_ratio/high_max": 0.11709614004939795, "clip_ratio/region_mean": 0.18323250487446785, "reward_total_mean": 0.5566011667251587, "reward_meter_mean": 0.8647017478942871, "reward_meter_std": 0.2624070942401886, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9960718154907227, "reward_repeat_soft_std": 0.0056772087700665, "reward_judge_quality_mean": 0.32624998688697815, "reward_judge_quality_std": 0.11350739002227783, "reward_total_composite_mean": 0.5566011667251587, "reward_total_composite_std": 0.36000367999076843} {"timestamp_utc": "2026-04-12T23:08:14Z", "mode": "train", "global_step": 375, "epoch": 0.03766951280763436, "loss": 0.1227, "grad_norm": 24.11614418029785, "learning_rate": 8.866666666666668e-06, "num_tokens": 723178.0, "completions/mean_length": 40.5, "completions/min_length": 34.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.5, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.7233703136444092, "rewards/meter/std": 0.32620927691459656, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.989782452583313, "rewards/repeat_soft/std": 0.027898427098989487, "rewards/judge_quality/mean": 0.7775000333786011, "rewards/judge_quality/std": 0.26921844482421875, "rewards/total_composite/mean": 0.8077448606491089, "rewards/total_composite/std": 0.18775489926338196, "reward": 0.8077448606491089, "reward_std": 0.18775489926338196, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16291044652462006, "sampling/sampling_logp_difference/max": 2.110621452331543, "sampling/importance_sampling_ratio/min": 0.12116264551877975, "sampling/importance_sampling_ratio/mean": 0.9809118509292603, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7950074672698975, "clip_ratio/low_mean": 0.03801478026434779, "clip_ratio/low_min": 0.03801478026434779, "clip_ratio/high_mean": 0.12500000558793545, "clip_ratio/high_max": 0.12500000558793545, "clip_ratio/region_mean": 0.16301478585228324, "reward_total_mean": 0.8077448606491089, "reward_meter_mean": 0.7233703136444092, "reward_meter_std": 0.32620927691459656, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.989782452583313, "reward_repeat_soft_std": 0.027898427098989487, "reward_judge_quality_mean": 0.7775000333786011, "reward_judge_quality_std": 0.26921844482421875, "reward_total_composite_mean": 0.8077448606491089, "reward_total_composite_std": 0.18775489926338196} {"timestamp_utc": "2026-04-12T23:08:20Z", "mode": "train", "global_step": 376, "epoch": 0.037769964841788045, "loss": 0.1025, "grad_norm": 13.997651100158691, "learning_rate": 8.863636363636365e-06, "num_tokens": 724687.0, "completions/mean_length": 30.625, "completions/min_length": 23.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.625, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9892204403877258, "rewards/meter/std": 0.00974253285676241, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9352542161941528, "rewards/repeat_soft/std": 0.04340337961912155, "rewards/judge_quality/mean": 0.42625004053115845, "rewards/judge_quality/std": 0.32693108916282654, "rewards/total_composite/mean": 0.8165495991706848, "rewards/total_composite/std": 0.10030784457921982, "reward": 0.8165495991706848, "reward_std": 0.10030784457921982, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1627560257911682, "sampling/sampling_logp_difference/max": 1.2901535034179688, "sampling/importance_sampling_ratio/min": 0.2752285301685333, "sampling/importance_sampling_ratio/mean": 1.0468833446502686, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6611389368772507, "clip_ratio/low_mean": 0.07815021555870771, "clip_ratio/low_min": 0.07815021555870771, "clip_ratio/high_mean": 0.07293127477169037, "clip_ratio/high_max": 0.07293127477169037, "clip_ratio/region_mean": 0.15108149033039808, "reward_total_mean": 0.8165495991706848, "reward_meter_mean": 0.9892204403877258, "reward_meter_std": 0.00974253285676241, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9352542161941528, "reward_repeat_soft_std": 0.04340337961912155, "reward_judge_quality_mean": 0.42625004053115845, "reward_judge_quality_std": 0.32693108916282654, "reward_total_composite_mean": 0.8165495991706848, "reward_total_composite_std": 0.10030784457921982} {"timestamp_utc": "2026-04-12T23:08:27Z", "mode": "train", "global_step": 377, "epoch": 0.03787041687594174, "loss": 0.0858, "grad_norm": 9.333416938781738, "learning_rate": 8.860606060606062e-06, "num_tokens": 726821.0, "completions/mean_length": 91.75, "completions/min_length": 61.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.75, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.8686516880989075, "rewards/meter/std": 0.12356135994195938, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9565036296844482, "rewards/repeat_soft/std": 0.04234939068555832, "rewards/judge_quality/mean": 0.45250001549720764, "rewards/judge_quality/std": 0.18100909888744354, "rewards/total_composite/mean": 0.7722936272621155, "rewards/total_composite/std": 0.09324721246957779, "reward": 0.7722936272621155, "reward_std": 0.09324721246957779, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17610488831996918, "sampling/sampling_logp_difference/max": 2.641946315765381, "sampling/importance_sampling_ratio/min": 0.16434520483016968, "sampling/importance_sampling_ratio/mean": 1.0328959226608276, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1450912803411484, "clip_ratio/low_mean": 0.09422864485532045, "clip_ratio/low_min": 0.09422864485532045, "clip_ratio/high_mean": 0.06953125074505806, "clip_ratio/high_max": 0.06953125074505806, "clip_ratio/region_mean": 0.1637598956003785, "reward_total_mean": 0.7722936272621155, "reward_meter_mean": 0.8686516880989075, "reward_meter_std": 0.12356135994195938, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9565036296844482, "reward_repeat_soft_std": 0.04234939068555832, "reward_judge_quality_mean": 0.45250001549720764, "reward_judge_quality_std": 0.18100909888744354, "reward_total_composite_mean": 0.7722936272621155, "reward_total_composite_std": 0.09324721246957779} {"timestamp_utc": "2026-04-12T23:08:35Z", "mode": "train", "global_step": 378, "epoch": 0.03797086891009543, "loss": 0.1182, "grad_norm": 7.397457122802734, "learning_rate": 8.857575757575758e-06, "num_tokens": 729256.0, "completions/mean_length": 115.375, "completions/min_length": 71.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.375, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.9043993949890137, "rewards/meter/std": 0.24792426824569702, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9133405089378357, "rewards/repeat_soft/std": 0.08183543384075165, "rewards/judge_quality/mean": 0.22374999523162842, "rewards/judge_quality/std": 0.12805551290512085, "rewards/total_composite/mean": 0.715438723564148, "rewards/total_composite/std": 0.12479452043771744, "reward": 0.715438723564148, "reward_std": 0.12479453533887863, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1550721824169159, "sampling/sampling_logp_difference/max": 1.256460189819336, "sampling/importance_sampling_ratio/min": 0.28465989232063293, "sampling/importance_sampling_ratio/mean": 1.0520564317703247, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0718109011650085, "clip_ratio/low_mean": 0.015384615398943424, "clip_ratio/low_min": 0.015384615398943424, "clip_ratio/high_mean": 0.10598900821059942, "clip_ratio/high_max": 0.10598900821059942, "clip_ratio/region_mean": 0.12137362360954285, "reward_total_mean": 0.715438723564148, "reward_meter_mean": 0.9043993949890137, "reward_meter_std": 0.24792426824569702, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9133405089378357, "reward_repeat_soft_std": 0.08183543384075165, "reward_judge_quality_mean": 0.22374999523162842, "reward_judge_quality_std": 0.12805551290512085, "reward_total_composite_mean": 0.715438723564148, "reward_total_composite_std": 0.12479452043771744} {"timestamp_utc": "2026-04-12T23:08:42Z", "mode": "train", "global_step": 379, "epoch": 0.03807132094424912, "loss": 0.0109, "grad_norm": 11.841979026794434, "learning_rate": 8.854545454545455e-06, "num_tokens": 731064.0, "completions/mean_length": 56.0, "completions/min_length": 51.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.0, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.5798560380935669, "rewards/meter/std": 0.44127988815307617, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9716790318489075, "rewards/repeat_soft/std": 0.03606419637799263, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.19334925711154938, "rewards/total_composite/mean": 0.6487281322479248, "rewards/total_composite/std": 0.18070362508296967, "reward": 0.6487281322479248, "reward_std": 0.18070361018180847, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16008812189102173, "sampling/sampling_logp_difference/max": 1.7227873802185059, "sampling/importance_sampling_ratio/min": 0.17856772243976593, "sampling/importance_sampling_ratio/mean": 1.0022099018096924, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1425939947366714, "clip_ratio/low_mean": 0.0617154510691762, "clip_ratio/low_min": 0.0617154510691762, "clip_ratio/high_mean": 0.07541858777403831, "clip_ratio/high_max": 0.07541858777403831, "clip_ratio/region_mean": 0.1371340388432145, "reward_total_mean": 0.6487281322479248, "reward_meter_mean": 0.5798560380935669, "reward_meter_std": 0.44127988815307617, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9716790318489075, "reward_repeat_soft_std": 0.03606419637799263, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.19334925711154938, "reward_total_composite_mean": 0.6487281322479248, "reward_total_composite_std": 0.18070362508296967} {"timestamp_utc": "2026-04-12T23:08:49Z", "mode": "train", "global_step": 380, "epoch": 0.03817177297840281, "loss": -0.0242, "grad_norm": 10.988927841186523, "learning_rate": 8.851515151515152e-06, "num_tokens": 732876.0, "completions/mean_length": 63.5, "completions/min_length": 47.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.5, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.4751718044281006, "rewards/meter/std": 0.3991663157939911, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.972125232219696, "rewards/repeat_soft/std": 0.03501483425498009, "rewards/judge_quality/mean": 0.45249998569488525, "rewards/judge_quality/std": 0.21224987506866455, "rewards/total_composite/mean": 0.5967898368835449, "rewards/total_composite/std": 0.15634436905384064, "reward": 0.5967898368835449, "reward_std": 0.15634436905384064, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1833529770374298, "sampling/sampling_logp_difference/max": 1.3986625671386719, "sampling/importance_sampling_ratio/min": 0.24692699313163757, "sampling/importance_sampling_ratio/mean": 1.014075517654419, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.669587068259716, "clip_ratio/low_mean": 0.07386924512684345, "clip_ratio/low_min": 0.07386924512684345, "clip_ratio/high_mean": 0.09172298200428486, "clip_ratio/high_max": 0.09172298200428486, "clip_ratio/region_mean": 0.1655922271311283, "reward_total_mean": 0.5967898368835449, "reward_meter_mean": 0.4751718044281006, "reward_meter_std": 0.3991663157939911, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.972125232219696, "reward_repeat_soft_std": 0.03501483425498009, "reward_judge_quality_mean": 0.45249998569488525, "reward_judge_quality_std": 0.21224987506866455, "reward_total_composite_mean": 0.5967898368835449, "reward_total_composite_std": 0.15634436905384064} {"timestamp_utc": "2026-04-12T23:08:55Z", "mode": "train", "global_step": 381, "epoch": 0.038272225012556504, "loss": 0.1009, "grad_norm": 20.435264587402344, "learning_rate": 8.84848484848485e-06, "num_tokens": 734442.0, "completions/mean_length": 35.75, "completions/min_length": 29.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.75, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.6975479125976562, "rewards/meter/std": 0.2978185713291168, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9804748892784119, "rewards/repeat_soft/std": 0.024515852332115173, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.8004440069198608, "rewards/total_composite/std": 0.13169269263744354, "reward": 0.8004440069198608, "reward_std": 0.13169269263744354, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16932888329029083, "sampling/sampling_logp_difference/max": 2.0124435424804688, "sampling/importance_sampling_ratio/min": 0.1336616724729538, "sampling/importance_sampling_ratio/mean": 1.0005627870559692, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.850868783891201, "clip_ratio/low_mean": 0.09076340682804585, "clip_ratio/low_min": 0.09076340682804585, "clip_ratio/high_mean": 0.10907041374593973, "clip_ratio/high_max": 0.10907041374593973, "clip_ratio/region_mean": 0.19983382057398558, "reward_total_mean": 0.8004440069198608, "reward_meter_mean": 0.6975479125976562, "reward_meter_std": 0.2978185713291168, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9804748892784119, "reward_repeat_soft_std": 0.024515852332115173, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.8004440069198608, "reward_total_composite_std": 0.13169269263744354} {"timestamp_utc": "2026-04-12T23:09:02Z", "mode": "train", "global_step": 382, "epoch": 0.0383726770467102, "loss": -0.0128, "grad_norm": 9.876946449279785, "learning_rate": 8.845454545454547e-06, "num_tokens": 736370.0, "completions/mean_length": 80.0, "completions/min_length": 71.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.0, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.4013100266456604, "rewards/meter/std": 0.38180917501449585, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9864307641983032, "rewards/repeat_soft/std": 0.018979214131832123, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.5424826145172119, "rewards/total_composite/std": 0.16715285181999207, "reward": 0.5424826145172119, "reward_std": 0.16715285181999207, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1893438994884491, "sampling/sampling_logp_difference/max": 2.0111989974975586, "sampling/importance_sampling_ratio/min": 0.13382811844348907, "sampling/importance_sampling_ratio/mean": 1.0197548866271973, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8972782045602798, "clip_ratio/low_mean": 0.07843270618468523, "clip_ratio/low_min": 0.07843270618468523, "clip_ratio/high_mean": 0.061568278819322586, "clip_ratio/high_max": 0.061568278819322586, "clip_ratio/region_mean": 0.14000098500400782, "reward_total_mean": 0.5424826145172119, "reward_meter_mean": 0.4013100266456604, "reward_meter_std": 0.38180917501449585, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9864307641983032, "reward_repeat_soft_std": 0.018979214131832123, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.5424826145172119, "reward_total_composite_std": 0.16715285181999207} {"timestamp_utc": "2026-04-12T23:09:09Z", "mode": "train", "global_step": 383, "epoch": 0.03847312908086389, "loss": 0.0953, "grad_norm": 6.58351469039917, "learning_rate": 8.842424242424244e-06, "num_tokens": 739118.0, "completions/mean_length": 140.5, "completions/min_length": 125.0, "completions/max_length": 161.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 140.5, "completions/min_terminated_length": 125.0, "completions/max_terminated_length": 161.0, "rewards/meter/mean": 0.6440650224685669, "rewards/meter/std": 0.3059301972389221, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9617048501968384, "rewards/repeat_soft/std": 0.04543507471680641, "rewards/judge_quality/mean": 0.26374998688697815, "rewards/judge_quality/std": 0.13373079895973206, "rewards/total_composite/mean": 0.6151247024536133, "rewards/total_composite/std": 0.16091257333755493, "reward": 0.6151247024536133, "reward_std": 0.16091257333755493, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17820453643798828, "sampling/sampling_logp_difference/max": 2.0355939865112305, "sampling/importance_sampling_ratio/min": 0.1306028962135315, "sampling/importance_sampling_ratio/mean": 1.034786581993103, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.201845556497574, "clip_ratio/low_mean": 0.056217764504253864, "clip_ratio/low_min": 0.056217764504253864, "clip_ratio/high_mean": 0.06110573234036565, "clip_ratio/high_max": 0.06110573234036565, "clip_ratio/region_mean": 0.11732349684461951, "reward_total_mean": 0.6151247024536133, "reward_meter_mean": 0.6440650224685669, "reward_meter_std": 0.3059301972389221, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9617048501968384, "reward_repeat_soft_std": 0.04543507471680641, "reward_judge_quality_mean": 0.26374998688697815, "reward_judge_quality_std": 0.13373079895973206, "reward_total_composite_mean": 0.6151247024536133, "reward_total_composite_std": 0.16091257333755493} {"timestamp_utc": "2026-04-12T23:09:16Z", "mode": "train", "global_step": 384, "epoch": 0.03857358111501758, "loss": 0.1167, "grad_norm": 10.995051383972168, "learning_rate": 8.83939393939394e-06, "num_tokens": 740828.0, "completions/mean_length": 54.75, "completions/min_length": 34.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.75, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.7842999696731567, "rewards/meter/std": 0.33811190724372864, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.971339762210846, "rewards/repeat_soft/std": 0.030454427003860474, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.18873640894889832, "rewards/total_composite/mean": 0.7583190202713013, "rewards/total_composite/std": 0.1823415756225586, "reward": 0.7583190202713013, "reward_std": 0.1823415607213974, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16792167723178864, "sampling/sampling_logp_difference/max": 1.4940643310546875, "sampling/importance_sampling_ratio/min": 0.22445853054523468, "sampling/importance_sampling_ratio/mean": 1.0586260557174683, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.133478820323944, "clip_ratio/low_mean": 0.062071364372968674, "clip_ratio/low_min": 0.062071364372968674, "clip_ratio/high_mean": 0.10551709681749344, "clip_ratio/high_max": 0.10551709681749344, "clip_ratio/region_mean": 0.1675884611904621, "reward_total_mean": 0.7583190202713013, "reward_meter_mean": 0.7842999696731567, "reward_meter_std": 0.33811190724372864, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.971339762210846, "reward_repeat_soft_std": 0.030454427003860474, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.18873640894889832, "reward_total_composite_mean": 0.7583190202713013, "reward_total_composite_std": 0.1823415756225586} {"timestamp_utc": "2026-04-12T23:09:22Z", "mode": "train", "global_step": 385, "epoch": 0.03867403314917127, "loss": 0.0157, "grad_norm": 35.86624526977539, "learning_rate": 8.836363636363637e-06, "num_tokens": 742155.0, "completions/mean_length": 18.875, "completions/min_length": 11.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 18.875, "completions/min_terminated_length": 11.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.4842010736465454, "rewards/meter/std": 0.25011277198791504, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.956183135509491, "rewards/repeat_soft/std": 0.017866725102066994, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5906338095664978, "rewards/total_composite/std": 0.11147178709506989, "reward": 0.5906338095664978, "reward_std": 0.1114717647433281, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12849506735801697, "sampling/sampling_logp_difference/max": 1.1139893531799316, "sampling/importance_sampling_ratio/min": 0.3282468616962433, "sampling/importance_sampling_ratio/mean": 1.0077884197235107, "sampling/importance_sampling_ratio/max": 1.9684346914291382, "entropy": 0.6291041318327188, "clip_ratio/low_mean": 0.04880952462553978, "clip_ratio/low_min": 0.04880952462553978, "clip_ratio/high_mean": 0.0935264932923019, "clip_ratio/high_max": 0.0935264932923019, "clip_ratio/region_mean": 0.14233601791784167, "reward_total_mean": 0.5906338095664978, "reward_meter_mean": 0.4842010736465454, "reward_meter_std": 0.25011277198791504, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.956183135509491, "reward_repeat_soft_std": 0.017866725102066994, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5906338095664978, "reward_total_composite_std": 0.11147178709506989} {"timestamp_utc": "2026-04-12T23:09:29Z", "mode": "train", "global_step": 386, "epoch": 0.038774485183324964, "loss": 0.0635, "grad_norm": 13.920676231384277, "learning_rate": 8.833333333333334e-06, "num_tokens": 743983.0, "completions/mean_length": 56.5, "completions/min_length": 46.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.5, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.8568496704101562, "rewards/meter/std": 0.22079578042030334, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9709453582763672, "rewards/repeat_soft/std": 0.018958723172545433, "rewards/judge_quality/mean": 0.5862500071525574, "rewards/judge_quality/std": 0.22984081506729126, "rewards/total_composite/mean": 0.8085519075393677, "rewards/total_composite/std": 0.14195625483989716, "reward": 0.8085519075393677, "reward_std": 0.14195623993873596, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16698652505874634, "sampling/sampling_logp_difference/max": 1.799781322479248, "sampling/importance_sampling_ratio/min": 0.1653350293636322, "sampling/importance_sampling_ratio/mean": 1.0304912328720093, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3985366821289062, "clip_ratio/low_mean": 0.04834475461393595, "clip_ratio/low_min": 0.04834475461393595, "clip_ratio/high_mean": 0.07443941663950682, "clip_ratio/high_max": 0.07443941663950682, "clip_ratio/region_mean": 0.12278417125344276, "reward_total_mean": 0.8085519075393677, "reward_meter_mean": 0.8568496704101562, "reward_meter_std": 0.22079578042030334, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9709453582763672, "reward_repeat_soft_std": 0.018958723172545433, "reward_judge_quality_mean": 0.5862500071525574, "reward_judge_quality_std": 0.22984081506729126, "reward_total_composite_mean": 0.8085519075393677, "reward_total_composite_std": 0.14195625483989716} {"timestamp_utc": "2026-04-12T23:09:37Z", "mode": "train", "global_step": 387, "epoch": 0.03887493721747865, "loss": 0.0209, "grad_norm": 11.725992202758789, "learning_rate": 8.830303030303031e-06, "num_tokens": 745798.0, "completions/mean_length": 60.875, "completions/min_length": 50.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.8483539819717407, "rewards/meter/std": 0.2913835644721985, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9988583922386169, "rewards/repeat_soft/std": 0.002009516814723611, "rewards/judge_quality/mean": 0.7112500667572021, "rewards/judge_quality/std": 0.18984487652778625, "rewards/total_composite/mean": 0.8450201749801636, "rewards/total_composite/std": 0.1527966558933258, "reward": 0.8450201749801636, "reward_std": 0.1527966409921646, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11517149209976196, "sampling/sampling_logp_difference/max": 1.2423820495605469, "sampling/importance_sampling_ratio/min": 0.28869572281837463, "sampling/importance_sampling_ratio/mean": 1.0124062299728394, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8555598668754101, "clip_ratio/low_mean": 0.034605263732373714, "clip_ratio/low_min": 0.034605263732373714, "clip_ratio/high_mean": 0.08383550774306059, "clip_ratio/high_max": 0.08383550774306059, "clip_ratio/region_mean": 0.1184407714754343, "reward_total_mean": 0.8450201749801636, "reward_meter_mean": 0.8483539819717407, "reward_meter_std": 0.2913835644721985, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9988583922386169, "reward_repeat_soft_std": 0.002009516814723611, "reward_judge_quality_mean": 0.7112500667572021, "reward_judge_quality_std": 0.18984487652778625, "reward_total_composite_mean": 0.8450201749801636, "reward_total_composite_std": 0.1527966558933258} {"timestamp_utc": "2026-04-12T23:09:44Z", "mode": "train", "global_step": 388, "epoch": 0.038975389251632346, "loss": 0.1315, "grad_norm": 11.183008193969727, "learning_rate": 8.827272727272727e-06, "num_tokens": 747403.0, "completions/mean_length": 50.625, "completions/min_length": 38.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.625, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.8789283037185669, "rewards/meter/std": 0.3148120641708374, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9929764866828918, "rewards/repeat_soft/std": 0.007867425680160522, "rewards/judge_quality/mean": 0.5175000429153442, "rewards/judge_quality/std": 0.21224986016750336, "rewards/total_composite/mean": 0.8000653982162476, "rewards/total_composite/std": 0.1618141233921051, "reward": 0.8000653982162476, "reward_std": 0.1618141233921051, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18400464951992035, "sampling/sampling_logp_difference/max": 1.3134326934814453, "sampling/importance_sampling_ratio/min": 0.2688954174518585, "sampling/importance_sampling_ratio/mean": 1.0462541580200195, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3126441687345505, "clip_ratio/low_mean": 0.029352466575801373, "clip_ratio/low_min": 0.029352466575801373, "clip_ratio/high_mean": 0.13120524678379297, "clip_ratio/high_max": 0.13120524678379297, "clip_ratio/region_mean": 0.16055771335959435, "reward_total_mean": 0.8000653982162476, "reward_meter_mean": 0.8789283037185669, "reward_meter_std": 0.3148120641708374, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9929764866828918, "reward_repeat_soft_std": 0.007867425680160522, "reward_judge_quality_mean": 0.5175000429153442, "reward_judge_quality_std": 0.21224986016750336, "reward_total_composite_mean": 0.8000653982162476, "reward_total_composite_std": 0.1618141233921051} {"timestamp_utc": "2026-04-12T23:09:51Z", "mode": "train", "global_step": 389, "epoch": 0.039075841285786034, "loss": -0.007, "grad_norm": 5.802651882171631, "learning_rate": 8.824242424242426e-06, "num_tokens": 749861.0, "completions/mean_length": 130.25, "completions/min_length": 124.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 130.25, "completions/min_terminated_length": 124.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.9587546586990356, "rewards/meter/std": 0.0954713448882103, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8044150471687317, "rewards/repeat_soft/std": 0.13126619160175323, "rewards/judge_quality/mean": 0.17125000059604645, "rewards/judge_quality/std": 0.11605878174304962, "rewards/total_composite/mean": 0.7132561206817627, "rewards/total_composite/std": 0.027461083605885506, "reward": 0.7132561206817627, "reward_std": 0.02746107615530491, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12348300963640213, "sampling/sampling_logp_difference/max": 1.8744815587997437, "sampling/importance_sampling_ratio/min": 0.15343450009822845, "sampling/importance_sampling_ratio/mean": 1.031293272972107, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.322613462805748, "clip_ratio/low_mean": 0.04839416313916445, "clip_ratio/low_min": 0.04839416313916445, "clip_ratio/high_mean": 0.050010460428893566, "clip_ratio/high_max": 0.050010460428893566, "clip_ratio/region_mean": 0.09840462356805801, "reward_total_mean": 0.7132561206817627, "reward_meter_mean": 0.9587546586990356, "reward_meter_std": 0.0954713448882103, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8044150471687317, "reward_repeat_soft_std": 0.13126619160175323, "reward_judge_quality_mean": 0.17125000059604645, "reward_judge_quality_std": 0.11605878174304962, "reward_total_composite_mean": 0.7132561206817627, "reward_total_composite_std": 0.027461083605885506} {"timestamp_utc": "2026-04-12T23:09:59Z", "mode": "train", "global_step": 390, "epoch": 0.03917629331993973, "loss": -0.0259, "grad_norm": 7.13739013671875, "learning_rate": 8.821212121212121e-06, "num_tokens": 752460.0, "completions/mean_length": 134.875, "completions/min_length": 110.0, "completions/max_length": 157.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 134.875, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.7454901933670044, "rewards/meter/std": 0.19132952392101288, "rewards/count_adherence/mean": 0.9166666269302368, "rewards/count_adherence/std": 0.0890870913863182, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8707361221313477, "rewards/repeat_soft/std": 0.05720118433237076, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.1249857097864151, "rewards/total_composite/mean": 0.5878694653511047, "rewards/total_composite/std": 0.2573975920677185, "reward": 0.5878694653511047, "reward_std": 0.2573975920677185, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13781459629535675, "sampling/sampling_logp_difference/max": 1.974466323852539, "sampling/importance_sampling_ratio/min": 0.1388353854417801, "sampling/importance_sampling_ratio/mean": 1.0092025995254517, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8724629506468773, "clip_ratio/low_mean": 0.04105558805167675, "clip_ratio/low_min": 0.04105558805167675, "clip_ratio/high_mean": 0.09390894509851933, "clip_ratio/high_max": 0.09390894509851933, "clip_ratio/region_mean": 0.13496453315019608, "reward_total_mean": 0.5878694653511047, "reward_meter_mean": 0.7454901933670044, "reward_meter_std": 0.19132952392101288, "reward_count_adherence_mean": 0.9166666269302368, "reward_count_adherence_std": 0.0890870913863182, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8707361221313477, "reward_repeat_soft_std": 0.05720118433237076, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.1249857097864151, "reward_total_composite_mean": 0.5878694653511047, "reward_total_composite_std": 0.2573975920677185} {"timestamp_utc": "2026-04-12T23:10:05Z", "mode": "train", "global_step": 391, "epoch": 0.03927674535409342, "loss": -0.0012, "grad_norm": 18.6479549407959, "learning_rate": 8.818181818181819e-06, "num_tokens": 753909.0, "completions/mean_length": 41.125, "completions/min_length": 31.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.125, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.8892362713813782, "rewards/meter/std": 0.13014128804206848, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9922200441360474, "rewards/repeat_soft/std": 0.010397223755717278, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.1865811049938202, "rewards/total_composite/mean": 0.8087533712387085, "rewards/total_composite/std": 0.0986696407198906, "reward": 0.8087533712387085, "reward_std": 0.09866963326931, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20341962575912476, "sampling/sampling_logp_difference/max": 1.508580207824707, "sampling/importance_sampling_ratio/min": 0.221223846077919, "sampling/importance_sampling_ratio/mean": 1.0174100399017334, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2314893305301666, "clip_ratio/low_mean": 0.09881163202226162, "clip_ratio/low_min": 0.09881163202226162, "clip_ratio/high_mean": 0.08860553428530693, "clip_ratio/high_max": 0.08860553428530693, "clip_ratio/region_mean": 0.18741716630756855, "reward_total_mean": 0.8087533712387085, "reward_meter_mean": 0.8892362713813782, "reward_meter_std": 0.13014128804206848, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9922200441360474, "reward_repeat_soft_std": 0.010397223755717278, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.1865811049938202, "reward_total_composite_mean": 0.8087533712387085, "reward_total_composite_std": 0.0986696407198906} {"timestamp_utc": "2026-04-12T23:10:12Z", "mode": "train", "global_step": 392, "epoch": 0.03937719738824711, "loss": -0.0186, "grad_norm": 10.528851509094238, "learning_rate": 8.815151515151516e-06, "num_tokens": 755963.0, "completions/mean_length": 80.75, "completions/min_length": 51.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.75, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.37706634402275085, "rewards/meter/std": 0.2786012887954712, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9650121331214905, "rewards/repeat_soft/std": 0.03934524953365326, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.1249857097864151, "rewards/total_composite/mean": 0.4574323892593384, "rewards/total_composite/std": 0.22447024285793304, "reward": 0.4574323892593384, "reward_std": 0.22447024285793304, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18355044722557068, "sampling/sampling_logp_difference/max": 1.234114646911621, "sampling/importance_sampling_ratio/min": 0.2910923659801483, "sampling/importance_sampling_ratio/mean": 1.0360018014907837, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.322569102048874, "clip_ratio/low_mean": 0.06873725261539221, "clip_ratio/low_min": 0.06873725261539221, "clip_ratio/high_mean": 0.10109396651387215, "clip_ratio/high_max": 0.10109396651387215, "clip_ratio/region_mean": 0.16983121912926435, "reward_total_mean": 0.4574323892593384, "reward_meter_mean": 0.37706634402275085, "reward_meter_std": 0.2786012887954712, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9650121331214905, "reward_repeat_soft_std": 0.03934524953365326, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.1249857097864151, "reward_total_composite_mean": 0.4574323892593384, "reward_total_composite_std": 0.22447024285793304} {"timestamp_utc": "2026-04-12T23:10:19Z", "mode": "train", "global_step": 393, "epoch": 0.039477649422400805, "loss": 0.0882, "grad_norm": 7.127695083618164, "learning_rate": 8.812121212121213e-06, "num_tokens": 758108.0, "completions/mean_length": 101.125, "completions/min_length": 77.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.125, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.7920228242874146, "rewards/meter/std": 0.36354485154151917, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8716015219688416, "rewards/repeat_soft/std": 0.1264854222536087, "rewards/judge_quality/mean": 0.3462499976158142, "rewards/judge_quality/std": 0.13721074163913727, "rewards/total_composite/mean": 0.6927579641342163, "rewards/total_composite/std": 0.15308870375156403, "reward": 0.6927579641342163, "reward_std": 0.15308870375156403, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15408554673194885, "sampling/sampling_logp_difference/max": 1.865671157836914, "sampling/importance_sampling_ratio/min": 0.15479227900505066, "sampling/importance_sampling_ratio/mean": 1.0220400094985962, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4170558378100395, "clip_ratio/low_mean": 0.028839003760367632, "clip_ratio/low_min": 0.028839003760367632, "clip_ratio/high_mean": 0.10011864267289639, "clip_ratio/high_max": 0.10011864267289639, "clip_ratio/region_mean": 0.12895764643326402, "reward_total_mean": 0.6927579641342163, "reward_meter_mean": 0.7920228242874146, "reward_meter_std": 0.36354485154151917, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8716015219688416, "reward_repeat_soft_std": 0.1264854222536087, "reward_judge_quality_mean": 0.3462499976158142, "reward_judge_quality_std": 0.13721074163913727, "reward_total_composite_mean": 0.6927579641342163, "reward_total_composite_std": 0.15308870375156403} {"timestamp_utc": "2026-04-12T23:10:26Z", "mode": "train", "global_step": 394, "epoch": 0.03957810145655449, "loss": -0.0476, "grad_norm": 11.833446502685547, "learning_rate": 8.809090909090909e-06, "num_tokens": 759855.0, "completions/mean_length": 51.375, "completions/min_length": 32.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.375, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.7738347053527832, "rewards/meter/std": 0.3227386176586151, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9819251298904419, "rewards/repeat_soft/std": 0.037453439086675644, "rewards/judge_quality/mean": 0.5900000333786011, "rewards/judge_quality/std": 0.27994900941848755, "rewards/total_composite/mean": 0.773418128490448, "rewards/total_composite/std": 0.17139944434165955, "reward": 0.773418128490448, "reward_std": 0.17139944434165955, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16987624764442444, "sampling/sampling_logp_difference/max": 2.227492332458496, "sampling/importance_sampling_ratio/min": 0.10779841244220734, "sampling/importance_sampling_ratio/mean": 1.0198628902435303, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.705855906009674, "clip_ratio/low_mean": 0.056625602301210165, "clip_ratio/low_min": 0.056625602301210165, "clip_ratio/high_mean": 0.09370767697691917, "clip_ratio/high_max": 0.09370767697691917, "clip_ratio/region_mean": 0.15033327927812934, "reward_total_mean": 0.773418128490448, "reward_meter_mean": 0.7738347053527832, "reward_meter_std": 0.3227386176586151, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9819251298904419, "reward_repeat_soft_std": 0.037453439086675644, "reward_judge_quality_mean": 0.5900000333786011, "reward_judge_quality_std": 0.27994900941848755, "reward_total_composite_mean": 0.773418128490448, "reward_total_composite_std": 0.17139944434165955} {"timestamp_utc": "2026-04-12T23:10:38Z", "mode": "train", "global_step": 395, "epoch": 0.03967855349070819, "loss": -0.2855, "grad_norm": 2.3390443325042725, "learning_rate": 8.806060606060608e-06, "num_tokens": 762543.0, "completions/mean_length": 226.0, "completions/min_length": 108.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 185.1428680419922, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 212.0, "rewards/meter/mean": 0.7722034454345703, "rewards/meter/std": 0.3584785759449005, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.1414213627576828, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9292647242546082, "rewards/repeat_soft/std": 0.0834198072552681, "rewards/judge_quality/mean": 0.16499999165534973, "rewards/judge_quality/std": 0.07708992809057236, "rewards/total_composite/mean": 0.5573177337646484, "rewards/total_composite/std": 0.2562350630760193, "reward": 0.5573177337646484, "reward_std": 0.2562350630760193, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1665474772453308, "sampling/sampling_logp_difference/max": 1.3652048110961914, "sampling/importance_sampling_ratio/min": 0.25532838702201843, "sampling/importance_sampling_ratio/mean": 1.0353467464447021, "sampling/importance_sampling_ratio/max": 1.987065076828003, "entropy": 2.498712196946144, "clip_ratio/low_mean": 0.02777777798473835, "clip_ratio/low_min": 0.02777777798473835, "clip_ratio/high_mean": 0.11881980579346418, "clip_ratio/high_max": 0.11881980579346418, "clip_ratio/region_mean": 0.14659758377820253, "reward_total_mean": 0.5573177337646484, "reward_meter_mean": 0.7722034454345703, "reward_meter_std": 0.3584785759449005, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.1414213627576828, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9292647242546082, "reward_repeat_soft_std": 0.0834198072552681, "reward_judge_quality_mean": 0.16499999165534973, "reward_judge_quality_std": 0.07708992809057236, "reward_total_composite_mean": 0.5573177337646484, "reward_total_composite_std": 0.2562350630760193} {"timestamp_utc": "2026-04-12T23:10:46Z", "mode": "train", "global_step": 396, "epoch": 0.039779005524861875, "loss": 0.0517, "grad_norm": 6.211075782775879, "learning_rate": 8.803030303030303e-06, "num_tokens": 765338.0, "completions/mean_length": 160.375, "completions/min_length": 114.0, "completions/max_length": 180.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 160.375, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 180.0, "rewards/meter/mean": 0.6524633169174194, "rewards/meter/std": 0.35910773277282715, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9474276304244995, "rewards/repeat_soft/std": 0.05586845800280571, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.6088512539863586, "rewards/total_composite/std": 0.16073913872241974, "reward": 0.6088512539863586, "reward_std": 0.16073912382125854, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17790557444095612, "sampling/sampling_logp_difference/max": 2.332524299621582, "sampling/importance_sampling_ratio/min": 0.09705045819282532, "sampling/importance_sampling_ratio/mean": 1.0308524370193481, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0336270332336426, "clip_ratio/low_mean": 0.059943494386971, "clip_ratio/low_min": 0.059943494386971, "clip_ratio/high_mean": 0.06167534925043583, "clip_ratio/high_max": 0.06167534925043583, "clip_ratio/region_mean": 0.12161884363740683, "reward_total_mean": 0.6088512539863586, "reward_meter_mean": 0.6524633169174194, "reward_meter_std": 0.35910773277282715, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9474276304244995, "reward_repeat_soft_std": 0.05586845800280571, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.6088512539863586, "reward_total_composite_std": 0.16073913872241974} {"timestamp_utc": "2026-04-12T23:10:53Z", "mode": "train", "global_step": 397, "epoch": 0.03987945755901557, "loss": 0.026, "grad_norm": 11.02540397644043, "learning_rate": 8.8e-06, "num_tokens": 767097.0, "completions/mean_length": 48.875, "completions/min_length": 34.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.875, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.6949447393417358, "rewards/meter/std": 0.35030439496040344, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9977872967720032, "rewards/repeat_soft/std": 0.0029111001640558243, "rewards/judge_quality/mean": 0.7150000333786011, "rewards/judge_quality/std": 0.2887411117553711, "rewards/total_composite/mean": 0.6806621551513672, "rewards/total_composite/std": 0.3449019491672516, "reward": 0.6806621551513672, "reward_std": 0.3449019491672516, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18597446382045746, "sampling/sampling_logp_difference/max": 1.7510528564453125, "sampling/importance_sampling_ratio/min": 0.17359107732772827, "sampling/importance_sampling_ratio/mean": 1.0403565168380737, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9460443705320358, "clip_ratio/low_mean": 0.051862807944417, "clip_ratio/low_min": 0.051862807944417, "clip_ratio/high_mean": 0.08661376126110554, "clip_ratio/high_max": 0.08661376126110554, "clip_ratio/region_mean": 0.13847656920552254, "reward_total_mean": 0.6806621551513672, "reward_meter_mean": 0.6949447393417358, "reward_meter_std": 0.35030439496040344, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9977872967720032, "reward_repeat_soft_std": 0.0029111001640558243, "reward_judge_quality_mean": 0.7150000333786011, "reward_judge_quality_std": 0.2887411117553711, "reward_total_composite_mean": 0.6806621551513672, "reward_total_composite_std": 0.3449019491672516} {"timestamp_utc": "2026-04-12T23:10:59Z", "mode": "train", "global_step": 398, "epoch": 0.039979909593169265, "loss": -0.0051, "grad_norm": 9.93213939666748, "learning_rate": 8.796969696969698e-06, "num_tokens": 768923.0, "completions/mean_length": 59.25, "completions/min_length": 53.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.25, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9148308038711548, "rewards/meter/std": 0.09792497754096985, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9720173478126526, "rewards/repeat_soft/std": 0.02282886393368244, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.8223756551742554, "rewards/total_composite/std": 0.06350842118263245, "reward": 0.8223756551742554, "reward_std": 0.06350842118263245, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1639508157968521, "sampling/sampling_logp_difference/max": 1.4908056259155273, "sampling/importance_sampling_ratio/min": 0.2941460609436035, "sampling/importance_sampling_ratio/mean": 1.0264570713043213, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.450992688536644, "clip_ratio/low_mean": 0.11204599589109421, "clip_ratio/low_min": 0.11204599589109421, "clip_ratio/high_mean": 0.029669540002942085, "clip_ratio/high_max": 0.029669540002942085, "clip_ratio/region_mean": 0.1417155358940363, "reward_total_mean": 0.8223756551742554, "reward_meter_mean": 0.9148308038711548, "reward_meter_std": 0.09792497754096985, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9720173478126526, "reward_repeat_soft_std": 0.02282886393368244, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.8223756551742554, "reward_total_composite_std": 0.06350842118263245} {"timestamp_utc": "2026-04-12T23:11:07Z", "mode": "train", "global_step": 399, "epoch": 0.04008036162732295, "loss": 0.0222, "grad_norm": 6.925746440887451, "learning_rate": 8.793939393939395e-06, "num_tokens": 771485.0, "completions/mean_length": 124.25, "completions/min_length": 81.0, "completions/max_length": 159.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 124.25, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 159.0, "rewards/meter/mean": 0.8623132705688477, "rewards/meter/std": 0.2926141023635864, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.914046585559845, "rewards/repeat_soft/std": 0.09695535153150558, "rewards/judge_quality/mean": 0.19625000655651093, "rewards/judge_quality/std": 0.11070391535758972, "rewards/total_composite/mean": 0.6789456605911255, "rewards/total_composite/std": 0.1260266900062561, "reward": 0.6789456605911255, "reward_std": 0.1260267049074173, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16981539130210876, "sampling/sampling_logp_difference/max": 1.9141631126403809, "sampling/importance_sampling_ratio/min": 0.14746519923210144, "sampling/importance_sampling_ratio/mean": 1.0482558012008667, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3993241488933563, "clip_ratio/low_mean": 0.03181103523820639, "clip_ratio/low_min": 0.03181103523820639, "clip_ratio/high_mean": 0.10007750894874334, "clip_ratio/high_max": 0.10007750894874334, "clip_ratio/region_mean": 0.13188854418694973, "reward_total_mean": 0.6789456605911255, "reward_meter_mean": 0.8623132705688477, "reward_meter_std": 0.2926141023635864, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.914046585559845, "reward_repeat_soft_std": 0.09695535153150558, "reward_judge_quality_mean": 0.19625000655651093, "reward_judge_quality_std": 0.11070391535758972, "reward_total_composite_mean": 0.6789456605911255, "reward_total_composite_std": 0.1260266900062561} {"timestamp_utc": "2026-04-12T23:11:13Z", "mode": "train", "global_step": 400, "epoch": 0.04018081366147665, "loss": 0.0475, "grad_norm": 14.895164489746094, "learning_rate": 8.790909090909092e-06, "num_tokens": 773055.0, "completions/mean_length": 44.25, "completions/min_length": 36.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.25, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.7154663801193237, "rewards/meter/std": 0.38724485039711, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.992277979850769, "rewards/repeat_soft/std": 0.008197621442377567, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.702812671661377, "rewards/total_composite/std": 0.1753060221672058, "reward": 0.702812671661377, "reward_std": 0.1753060221672058, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18309256434440613, "sampling/sampling_logp_difference/max": 1.5150785446166992, "sampling/importance_sampling_ratio/min": 0.21979093551635742, "sampling/importance_sampling_ratio/mean": 1.026138186454773, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7868029177188873, "clip_ratio/low_mean": 0.04289215803146362, "clip_ratio/low_min": 0.04289215803146362, "clip_ratio/high_mean": 0.10702694393694401, "clip_ratio/high_max": 0.10702694393694401, "clip_ratio/region_mean": 0.14991910196840763, "reward_total_mean": 0.702812671661377, "reward_meter_mean": 0.7154663801193237, "reward_meter_std": 0.38724485039711, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.992277979850769, "reward_repeat_soft_std": 0.008197621442377567, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.702812671661377, "reward_total_composite_std": 0.1753060221672058} {"timestamp_utc": "2026-04-12T23:12:37Z", "mode": "eval", "global_step": 400, "epoch": 0.04018081366147665, "eval_loss": NaN, "eval_runtime": 83.2057, "eval_samples_per_second": 0.961, "eval_steps_per_second": 0.12, "eval_num_tokens": 773055.0, "eval_completions/mean_length": 107.7375, "eval_completions/min_length": 31.7, "eval_completions/max_length": 299.0, "eval_completions/clipped_ratio": 0.05, "eval_completions/mean_terminated_length": 86.81607360839844, "eval_completions/min_terminated_length": 31.7, "eval_completions/max_terminated_length": 170.7, "eval_rewards/meter/mean": 0.5810104548931122, "eval_rewards/meter/std": 0.38152158856391905, "eval_rewards/count_adherence/mean": 0.9095833241939545, "eval_rewards/count_adherence/std": 0.1434095051139593, "eval_rewards/hard_gate/mean": 0.9375, "eval_rewards/hard_gate/std": 0.1767766922712326, "eval_rewards/repeat_soft/mean": 0.9517585337162018, "eval_rewards/repeat_soft/std": 0.07104268427938223, "eval_rewards/judge_quality/mean": 0.4091250002384186, "eval_rewards/judge_quality/std": 0.20390536040067672, "eval_rewards/total_composite/mean": 0.5834895074367523, "eval_rewards/total_composite/std": 0.2427708923816681, "eval_reward": 0.5834895074367523, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.13129957914352416, "eval_sampling/sampling_logp_difference/max": 1.1551784992218017, "eval_sampling/importance_sampling_ratio/min": 0.3216679602861404, "eval_sampling/importance_sampling_ratio/mean": 1.036140215396881, "eval_sampling/importance_sampling_ratio/max": 1.5824068903923034, "eval_entropy": 1.9290939688682556, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5834895074367523, "eval_reward_meter_mean": 0.5810104548931122, "eval_reward_meter_std": 0.38152158856391905, "eval_reward_count_adherence_mean": 0.9095833241939545, "eval_reward_count_adherence_std": 0.1434095051139593, "eval_reward_hard_gate_mean": 0.9375, "eval_reward_hard_gate_std": 0.1767766922712326, "eval_reward_repeat_soft_mean": 0.9517585337162018, "eval_reward_repeat_soft_std": 0.07104268427938223, "eval_reward_judge_quality_mean": 0.4091250002384186, "eval_reward_judge_quality_std": 0.20390536040067672, "eval_reward_total_composite_mean": 0.5834895074367523, "eval_reward_total_composite_std": 0.2427708923816681} {"timestamp_utc": "2026-04-12T23:12:46Z", "mode": "train", "global_step": 401, "epoch": 0.040281265695630335, "loss": 0.0166, "grad_norm": 8.335186958312988, "learning_rate": 8.787878787878788e-06, "num_tokens": 775177.0, "completions/mean_length": 85.25, "completions/min_length": 70.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.25, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.46360456943511963, "rewards/meter/std": 0.27548080682754517, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9579570293426514, "rewards/repeat_soft/std": 0.04774823039770126, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.6057302951812744, "rewards/total_composite/std": 0.16374103724956512, "reward": 0.6057302951812744, "reward_std": 0.16374102234840393, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12874223291873932, "sampling/sampling_logp_difference/max": 1.6254808902740479, "sampling/importance_sampling_ratio/min": 0.19681701064109802, "sampling/importance_sampling_ratio/mean": 1.003516435623169, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8052221685647964, "clip_ratio/low_mean": 0.07227678876370192, "clip_ratio/low_min": 0.07227678876370192, "clip_ratio/high_mean": 0.05883005913347006, "clip_ratio/high_max": 0.05883005913347006, "clip_ratio/region_mean": 0.13110684789717197, "reward_total_mean": 0.6057302951812744, "reward_meter_mean": 0.46360456943511963, "reward_meter_std": 0.27548080682754517, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9579570293426514, "reward_repeat_soft_std": 0.04774823039770126, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.6057302951812744, "reward_total_composite_std": 0.16374103724956512} {"timestamp_utc": "2026-04-12T23:12:54Z", "mode": "train", "global_step": 402, "epoch": 0.04038171772978403, "loss": 0.1342, "grad_norm": 17.314476013183594, "learning_rate": 8.784848484848487e-06, "num_tokens": 776703.0, "completions/mean_length": 30.75, "completions/min_length": 19.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.75, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.8092689514160156, "rewards/meter/std": 0.3480612635612488, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.947524905204773, "rewards/repeat_soft/std": 0.022066960111260414, "rewards/judge_quality/mean": 0.29249998927116394, "rewards/judge_quality/std": 0.12150837481021881, "rewards/total_composite/mean": 0.6779235601425171, "rewards/total_composite/std": 0.16187335550785065, "reward": 0.6779235601425171, "reward_std": 0.16187335550785065, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17064008116722107, "sampling/sampling_logp_difference/max": 1.1062941551208496, "sampling/importance_sampling_ratio/min": 0.33078253269195557, "sampling/importance_sampling_ratio/mean": 1.014972448348999, "sampling/importance_sampling_ratio/max": 1.8557288646697998, "entropy": 1.7295797914266586, "clip_ratio/low_mean": 0.02222222276031971, "clip_ratio/low_min": 0.02222222276031971, "clip_ratio/high_mean": 0.0949717927724123, "clip_ratio/high_max": 0.0949717927724123, "clip_ratio/region_mean": 0.11719401553273201, "reward_total_mean": 0.6779235601425171, "reward_meter_mean": 0.8092689514160156, "reward_meter_std": 0.3480612635612488, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.947524905204773, "reward_repeat_soft_std": 0.022066960111260414, "reward_judge_quality_mean": 0.29249998927116394, "reward_judge_quality_std": 0.12150837481021881, "reward_total_composite_mean": 0.6779235601425171, "reward_total_composite_std": 0.16187335550785065} {"timestamp_utc": "2026-04-12T23:13:01Z", "mode": "train", "global_step": 403, "epoch": 0.04048216976393772, "loss": 0.1812, "grad_norm": 20.47299575805664, "learning_rate": 8.781818181818182e-06, "num_tokens": 778189.0, "completions/mean_length": 25.75, "completions/min_length": 17.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.75, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.928257405757904, "rewards/meter/std": 0.09230384230613708, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.4629100561141968, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9673579931259155, "rewards/repeat_soft/std": 0.013197989203035831, "rewards/judge_quality/mean": 0.6812499761581421, "rewards/judge_quality/std": 0.25542333722114563, "rewards/total_composite/mean": 0.8313266038894653, "rewards/total_composite/std": 0.15258866548538208, "reward": 0.8313266038894653, "reward_std": 0.1525886505842209, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1677527278661728, "sampling/sampling_logp_difference/max": 1.3376272916793823, "sampling/importance_sampling_ratio/min": 0.2624676823616028, "sampling/importance_sampling_ratio/mean": 1.0065242052078247, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.553636372089386, "clip_ratio/low_mean": 0.0467640720307827, "clip_ratio/low_min": 0.0467640720307827, "clip_ratio/high_mean": 0.09339986089617014, "clip_ratio/high_max": 0.09339986089617014, "clip_ratio/region_mean": 0.14016393292695284, "reward_total_mean": 0.8313266038894653, "reward_meter_mean": 0.928257405757904, "reward_meter_std": 0.09230384230613708, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.4629100561141968, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9673579931259155, "reward_repeat_soft_std": 0.013197989203035831, "reward_judge_quality_mean": 0.6812499761581421, "reward_judge_quality_std": 0.25542333722114563, "reward_total_composite_mean": 0.8313266038894653, "reward_total_composite_std": 0.15258866548538208} {"timestamp_utc": "2026-04-12T23:13:13Z", "mode": "train", "global_step": 404, "epoch": 0.04058262179809141, "loss": -0.2008, "grad_norm": 2.0813241004943848, "learning_rate": 8.77878787878788e-06, "num_tokens": 780600.0, "completions/mean_length": 167.375, "completions/min_length": 93.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 118.14286041259766, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.8614681363105774, "rewards/meter/std": 0.28220465779304504, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8987991809844971, "rewards/repeat_soft/std": 0.18123045563697815, "rewards/judge_quality/mean": 0.23874998092651367, "rewards/judge_quality/std": 0.16504868865013123, "rewards/total_composite/mean": 0.6486514806747437, "rewards/total_composite/std": 0.269854873418808, "reward": 0.6486514806747437, "reward_std": 0.26985490322113037, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18065668642520905, "sampling/sampling_logp_difference/max": 1.6054878234863281, "sampling/importance_sampling_ratio/min": 0.20079158246517181, "sampling/importance_sampling_ratio/mean": 1.0399219989776611, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0141394957900047, "clip_ratio/low_mean": 0.008395522832870483, "clip_ratio/low_min": 0.008395522832870483, "clip_ratio/high_mean": 0.13276397250592709, "clip_ratio/high_max": 0.13276397250592709, "clip_ratio/region_mean": 0.14115949533879757, "reward_total_mean": 0.6486514806747437, "reward_meter_mean": 0.8614681363105774, "reward_meter_std": 0.28220465779304504, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8987991809844971, "reward_repeat_soft_std": 0.18123045563697815, "reward_judge_quality_mean": 0.23874998092651367, "reward_judge_quality_std": 0.16504868865013123, "reward_total_composite_mean": 0.6486514806747437, "reward_total_composite_std": 0.269854873418808} {"timestamp_utc": "2026-04-12T23:13:24Z", "mode": "train", "global_step": 405, "epoch": 0.040683073832245106, "loss": -0.077, "grad_norm": 3.9539730548858643, "learning_rate": 8.775757575757577e-06, "num_tokens": 782760.0, "completions/mean_length": 150.0, "completions/min_length": 69.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 98.28572082519531, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.5335673093795776, "rewards/meter/std": 0.30695945024490356, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9802958369255066, "rewards/repeat_soft/std": 0.026636289432644844, "rewards/judge_quality/mean": 0.35374999046325684, "rewards/judge_quality/std": 0.26348966360092163, "rewards/total_composite/mean": 0.5274440050125122, "rewards/total_composite/std": 0.2874928116798401, "reward": 0.5274440050125122, "reward_std": 0.2874927818775177, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1880018264055252, "sampling/sampling_logp_difference/max": 1.7351384162902832, "sampling/importance_sampling_ratio/min": 0.17637579143047333, "sampling/importance_sampling_ratio/mean": 1.039862036705017, "sampling/importance_sampling_ratio/max": 1.879452109336853, "entropy": 1.8812029510736465, "clip_ratio/low_mean": 0.057575950399041176, "clip_ratio/low_min": 0.057575950399041176, "clip_ratio/high_mean": 0.08322884421795607, "clip_ratio/high_max": 0.08322884421795607, "clip_ratio/region_mean": 0.14080479461699724, "reward_total_mean": 0.5274440050125122, "reward_meter_mean": 0.5335673093795776, "reward_meter_std": 0.30695945024490356, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9802958369255066, "reward_repeat_soft_std": 0.026636289432644844, "reward_judge_quality_mean": 0.35374999046325684, "reward_judge_quality_std": 0.26348966360092163, "reward_total_composite_mean": 0.5274440050125122, "reward_total_composite_std": 0.2874928116798401} {"timestamp_utc": "2026-04-12T23:13:31Z", "mode": "train", "global_step": 406, "epoch": 0.040783525866398794, "loss": 0.0061, "grad_norm": 11.950566291809082, "learning_rate": 8.772727272727274e-06, "num_tokens": 784448.0, "completions/mean_length": 56.0, "completions/min_length": 43.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.0, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9490143656730652, "rewards/meter/std": 0.07202045619487762, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9502244591712952, "rewards/repeat_soft/std": 0.09206971526145935, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.0843462198972702, "rewards/total_composite/mean": 0.7875789403915405, "rewards/total_composite/std": 0.038430094718933105, "reward": 0.7875789403915405, "reward_std": 0.03843009099364281, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16331587731838226, "sampling/sampling_logp_difference/max": 0.9833202362060547, "sampling/importance_sampling_ratio/min": 0.3740670680999756, "sampling/importance_sampling_ratio/mean": 1.0354595184326172, "sampling/importance_sampling_ratio/max": 1.9068964719772339, "entropy": 1.768060326576233, "clip_ratio/low_mean": 0.046498351730406284, "clip_ratio/low_min": 0.046498351730406284, "clip_ratio/high_mean": 0.12079948745667934, "clip_ratio/high_max": 0.12079948745667934, "clip_ratio/region_mean": 0.16729783918708563, "reward_total_mean": 0.7875789403915405, "reward_meter_mean": 0.9490143656730652, "reward_meter_std": 0.07202045619487762, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9502244591712952, "reward_repeat_soft_std": 0.09206971526145935, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.0843462198972702, "reward_total_composite_mean": 0.7875789403915405, "reward_total_composite_std": 0.038430094718933105} {"timestamp_utc": "2026-04-12T23:13:37Z", "mode": "train", "global_step": 407, "epoch": 0.04088397790055249, "loss": -0.0121, "grad_norm": 12.772229194641113, "learning_rate": 8.76969696969697e-06, "num_tokens": 786118.0, "completions/mean_length": 38.75, "completions/min_length": 29.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.75, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.6010382771492004, "rewards/meter/std": 0.44268232583999634, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9933664798736572, "rewards/repeat_soft/std": 0.008879461325705051, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.7395538687705994, "rewards/total_composite/std": 0.2555656433105469, "reward": 0.7395538687705994, "reward_std": 0.2555656433105469, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15382200479507446, "sampling/sampling_logp_difference/max": 1.1537742614746094, "sampling/importance_sampling_ratio/min": 0.3154439628124237, "sampling/importance_sampling_ratio/mean": 1.029845952987671, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3676116615533829, "clip_ratio/low_mean": 0.06574353482574224, "clip_ratio/low_min": 0.06574353482574224, "clip_ratio/high_mean": 0.06009569391608238, "clip_ratio/high_max": 0.06009569391608238, "clip_ratio/region_mean": 0.12583922874182463, "reward_total_mean": 0.7395538687705994, "reward_meter_mean": 0.6010382771492004, "reward_meter_std": 0.44268232583999634, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9933664798736572, "reward_repeat_soft_std": 0.008879461325705051, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.7395538687705994, "reward_total_composite_std": 0.2555656433105469} {"timestamp_utc": "2026-04-12T23:13:44Z", "mode": "train", "global_step": 408, "epoch": 0.040984429934706176, "loss": 0.1814, "grad_norm": 25.650049209594727, "learning_rate": 8.766666666666669e-06, "num_tokens": 787800.0, "completions/mean_length": 35.25, "completions/min_length": 21.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.25, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.4753566384315491, "rewards/meter/std": 0.38857781887054443, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9794173240661621, "rewards/repeat_soft/std": 0.02277480438351631, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.1940544992685318, "rewards/total_composite/mean": 0.5919772386550903, "rewards/total_composite/std": 0.22380661964416504, "reward": 0.5919772386550903, "reward_std": 0.22380661964416504, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20901210606098175, "sampling/sampling_logp_difference/max": 1.7618727684020996, "sampling/importance_sampling_ratio/min": 0.17172296345233917, "sampling/importance_sampling_ratio/mean": 1.0847902297973633, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.134936213493347, "clip_ratio/low_mean": 0.12923871539533138, "clip_ratio/low_min": 0.12923871539533138, "clip_ratio/high_mean": 0.060591437853872776, "clip_ratio/high_max": 0.060591437853872776, "clip_ratio/region_mean": 0.18983015324920416, "reward_total_mean": 0.5919772386550903, "reward_meter_mean": 0.4753566384315491, "reward_meter_std": 0.38857781887054443, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9794173240661621, "reward_repeat_soft_std": 0.02277480438351631, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.1940544992685318, "reward_total_composite_mean": 0.5919772386550903, "reward_total_composite_std": 0.22380661964416504} {"timestamp_utc": "2026-04-12T23:13:52Z", "mode": "train", "global_step": 409, "epoch": 0.04108488196885987, "loss": 0.0372, "grad_norm": 7.020883083343506, "learning_rate": 8.763636363636364e-06, "num_tokens": 790321.0, "completions/mean_length": 128.125, "completions/min_length": 117.0, "completions/max_length": 152.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.125, "completions/min_terminated_length": 117.0, "completions/max_terminated_length": 152.0, "rewards/meter/mean": 0.6893325448036194, "rewards/meter/std": 0.3986375033855438, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8790450096130371, "rewards/repeat_soft/std": 0.11994105577468872, "rewards/judge_quality/mean": 0.33500000834465027, "rewards/judge_quality/std": 0.19992856681346893, "rewards/total_composite/mean": 0.643916666507721, "rewards/total_composite/std": 0.15544717013835907, "reward": 0.643916666507721, "reward_std": 0.15544715523719788, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12568385899066925, "sampling/sampling_logp_difference/max": 1.7258915901184082, "sampling/importance_sampling_ratio/min": 0.17801426351070404, "sampling/importance_sampling_ratio/mean": 1.0223184823989868, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9791223183274269, "clip_ratio/low_mean": 0.05755038093775511, "clip_ratio/low_min": 0.05755038093775511, "clip_ratio/high_mean": 0.05657648574560881, "clip_ratio/high_max": 0.05657648574560881, "clip_ratio/region_mean": 0.11412686668336391, "reward_total_mean": 0.643916666507721, "reward_meter_mean": 0.6893325448036194, "reward_meter_std": 0.3986375033855438, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8790450096130371, "reward_repeat_soft_std": 0.11994105577468872, "reward_judge_quality_mean": 0.33500000834465027, "reward_judge_quality_std": 0.19992856681346893, "reward_total_composite_mean": 0.643916666507721, "reward_total_composite_std": 0.15544717013835907} {"timestamp_utc": "2026-04-12T23:13:58Z", "mode": "train", "global_step": 410, "epoch": 0.04118533400301356, "loss": 0.0319, "grad_norm": 14.928460121154785, "learning_rate": 8.760606060606061e-06, "num_tokens": 791971.0, "completions/mean_length": 41.25, "completions/min_length": 31.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.25, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.7796938419342041, "rewards/meter/std": 0.27412301301956177, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9927184581756592, "rewards/repeat_soft/std": 0.011512663215398788, "rewards/judge_quality/mean": 0.8612500429153442, "rewards/judge_quality/std": 0.16617010533809662, "rewards/total_composite/mean": 0.8585090637207031, "rewards/total_composite/std": 0.11834528297185898, "reward": 0.8585090637207031, "reward_std": 0.11834526062011719, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1366901397705078, "sampling/sampling_logp_difference/max": 1.2815666198730469, "sampling/importance_sampling_ratio/min": 0.27760207653045654, "sampling/importance_sampling_ratio/mean": 1.0082085132598877, "sampling/importance_sampling_ratio/max": 1.8112285137176514, "entropy": 1.0979838892817497, "clip_ratio/low_mean": 0.07824705727398396, "clip_ratio/low_min": 0.07824705727398396, "clip_ratio/high_mean": 0.0745284054428339, "clip_ratio/high_max": 0.0745284054428339, "clip_ratio/region_mean": 0.15277546271681786, "reward_total_mean": 0.8585090637207031, "reward_meter_mean": 0.7796938419342041, "reward_meter_std": 0.27412301301956177, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9927184581756592, "reward_repeat_soft_std": 0.011512663215398788, "reward_judge_quality_mean": 0.8612500429153442, "reward_judge_quality_std": 0.16617010533809662, "reward_total_composite_mean": 0.8585090637207031, "reward_total_composite_std": 0.11834528297185898} {"timestamp_utc": "2026-04-12T23:14:04Z", "mode": "train", "global_step": 411, "epoch": 0.04128578603716725, "loss": -0.0641, "grad_norm": 16.129539489746094, "learning_rate": 8.757575757575759e-06, "num_tokens": 793591.0, "completions/mean_length": 42.5, "completions/min_length": 30.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.5, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.8029767274856567, "rewards/meter/std": 0.2548975646495819, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9670915603637695, "rewards/repeat_soft/std": 0.03096822276711464, "rewards/judge_quality/mean": 0.5012500286102295, "rewards/judge_quality/std": 0.16974246501922607, "rewards/total_composite/mean": 0.6903079152107239, "rewards/total_composite/std": 0.2985648512840271, "reward": 0.6903079152107239, "reward_std": 0.2985648214817047, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12839943170547485, "sampling/sampling_logp_difference/max": 1.3263778686523438, "sampling/importance_sampling_ratio/min": 0.26543697714805603, "sampling/importance_sampling_ratio/mean": 1.0291063785552979, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5071376860141754, "clip_ratio/low_mean": 0.02967479731887579, "clip_ratio/low_min": 0.02967479731887579, "clip_ratio/high_mean": 0.09930349420756102, "clip_ratio/high_max": 0.09930349420756102, "clip_ratio/region_mean": 0.1289782915264368, "reward_total_mean": 0.6903079152107239, "reward_meter_mean": 0.8029767274856567, "reward_meter_std": 0.2548975646495819, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9670915603637695, "reward_repeat_soft_std": 0.03096822276711464, "reward_judge_quality_mean": 0.5012500286102295, "reward_judge_quality_std": 0.16974246501922607, "reward_total_composite_mean": 0.6903079152107239, "reward_total_composite_std": 0.2985648512840271} {"timestamp_utc": "2026-04-12T23:14:11Z", "mode": "train", "global_step": 412, "epoch": 0.04138623807132094, "loss": 0.2847, "grad_norm": 13.445656776428223, "learning_rate": 8.754545454545456e-06, "num_tokens": 795060.0, "completions/mean_length": 43.625, "completions/min_length": 27.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.625, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.6817044019699097, "rewards/meter/std": 0.38767337799072266, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9685162901878357, "rewards/repeat_soft/std": 0.0650566890835762, "rewards/judge_quality/mean": 0.543749988079071, "rewards/judge_quality/std": 0.17816025018692017, "rewards/total_composite/mean": 0.7073686122894287, "rewards/total_composite/std": 0.20675984025001526, "reward": 0.7073686122894287, "reward_std": 0.20675982534885406, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1512790471315384, "sampling/sampling_logp_difference/max": 1.4083738327026367, "sampling/importance_sampling_ratio/min": 0.24454063177108765, "sampling/importance_sampling_ratio/mean": 1.0067201852798462, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9220719784498215, "clip_ratio/low_mean": 0.015801096335053444, "clip_ratio/low_min": 0.015801096335053444, "clip_ratio/high_mean": 0.11759095685556531, "clip_ratio/high_max": 0.11759095685556531, "clip_ratio/region_mean": 0.13339205319061875, "reward_total_mean": 0.7073686122894287, "reward_meter_mean": 0.6817044019699097, "reward_meter_std": 0.38767337799072266, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9685162901878357, "reward_repeat_soft_std": 0.0650566890835762, "reward_judge_quality_mean": 0.543749988079071, "reward_judge_quality_std": 0.17816025018692017, "reward_total_composite_mean": 0.7073686122894287, "reward_total_composite_std": 0.20675984025001526} {"timestamp_utc": "2026-04-12T23:14:18Z", "mode": "train", "global_step": 413, "epoch": 0.041486690105474636, "loss": 0.1573, "grad_norm": 10.743023872375488, "learning_rate": 8.751515151515151e-06, "num_tokens": 797053.0, "completions/mean_length": 73.125, "completions/min_length": 51.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.125, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.9447565078735352, "rewards/meter/std": 0.06909362226724625, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9045273661613464, "rewards/repeat_soft/std": 0.10507004708051682, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.8040931224822998, "rewards/total_composite/std": 0.07532332092523575, "reward": 0.8040931224822998, "reward_std": 0.07532332092523575, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1649683564901352, "sampling/sampling_logp_difference/max": 4.701249122619629, "sampling/importance_sampling_ratio/min": 0.2333166003227234, "sampling/importance_sampling_ratio/mean": 1.0188426971435547, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4624207392334938, "clip_ratio/low_mean": 0.056584584061056376, "clip_ratio/low_min": 0.056584584061056376, "clip_ratio/high_mean": 0.09021655097603798, "clip_ratio/high_max": 0.09021655097603798, "clip_ratio/region_mean": 0.14680113503709435, "reward_total_mean": 0.8040931224822998, "reward_meter_mean": 0.9447565078735352, "reward_meter_std": 0.06909362226724625, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9045273661613464, "reward_repeat_soft_std": 0.10507004708051682, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.8040931224822998, "reward_total_composite_std": 0.07532332092523575} {"timestamp_utc": "2026-04-12T23:14:25Z", "mode": "train", "global_step": 414, "epoch": 0.04158714213962833, "loss": 0.0186, "grad_norm": 16.562807083129883, "learning_rate": 8.748484848484849e-06, "num_tokens": 798518.0, "completions/mean_length": 28.125, "completions/min_length": 20.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.125, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.5023785829544067, "rewards/meter/std": 0.30153220891952515, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9520687460899353, "rewards/repeat_soft/std": 0.02666119858622551, "rewards/judge_quality/mean": 0.39750000834465027, "rewards/judge_quality/std": 0.10110107809305191, "rewards/total_composite/mean": 0.5905272364616394, "rewards/total_composite/std": 0.14232338964939117, "reward": 0.5905272364616394, "reward_std": 0.14232338964939117, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1984531581401825, "sampling/sampling_logp_difference/max": 1.2772550582885742, "sampling/importance_sampling_ratio/min": 0.2788015305995941, "sampling/importance_sampling_ratio/mean": 1.0257309675216675, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9693736284971237, "clip_ratio/low_mean": 0.08987471088767052, "clip_ratio/low_min": 0.08987471088767052, "clip_ratio/high_mean": 0.10175729636102915, "clip_ratio/high_max": 0.10175729636102915, "clip_ratio/region_mean": 0.19163200724869967, "reward_total_mean": 0.5905272364616394, "reward_meter_mean": 0.5023785829544067, "reward_meter_std": 0.30153220891952515, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9520687460899353, "reward_repeat_soft_std": 0.02666119858622551, "reward_judge_quality_mean": 0.39750000834465027, "reward_judge_quality_std": 0.10110107809305191, "reward_total_composite_mean": 0.5905272364616394, "reward_total_composite_std": 0.14232338964939117} {"timestamp_utc": "2026-04-12T23:14:31Z", "mode": "train", "global_step": 415, "epoch": 0.04168759417378202, "loss": -0.007, "grad_norm": 23.003585815429688, "learning_rate": 8.745454545454546e-06, "num_tokens": 800285.0, "completions/mean_length": 56.875, "completions/min_length": 44.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.875, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.6583911180496216, "rewards/meter/std": 0.4006628096103668, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9652466177940369, "rewards/repeat_soft/std": 0.04198618233203888, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.6875506639480591, "rewards/total_composite/std": 0.1530541479587555, "reward": 0.6875506639480591, "reward_std": 0.1530541330575943, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19211675226688385, "sampling/sampling_logp_difference/max": 1.6794543266296387, "sampling/importance_sampling_ratio/min": 0.1864756941795349, "sampling/importance_sampling_ratio/mean": 1.0060641765594482, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1600653305649757, "clip_ratio/low_mean": 0.06835648510605097, "clip_ratio/low_min": 0.06835648510605097, "clip_ratio/high_mean": 0.11648799106478691, "clip_ratio/high_max": 0.11648799106478691, "clip_ratio/region_mean": 0.18484447617083788, "reward_total_mean": 0.6875506639480591, "reward_meter_mean": 0.6583911180496216, "reward_meter_std": 0.4006628096103668, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9652466177940369, "reward_repeat_soft_std": 0.04198618233203888, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.6875506639480591, "reward_total_composite_std": 0.1530541479587555} {"timestamp_utc": "2026-04-12T23:14:38Z", "mode": "train", "global_step": 416, "epoch": 0.04178804620793571, "loss": 0.1085, "grad_norm": 8.683070182800293, "learning_rate": 8.742424242424243e-06, "num_tokens": 801900.0, "completions/mean_length": 52.875, "completions/min_length": 35.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.875, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.992560625076294, "rewards/meter/std": 0.00855978112667799, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9587268829345703, "rewards/repeat_soft/std": 0.050369251519441605, "rewards/judge_quality/mean": 0.39375001192092896, "rewards/judge_quality/std": 0.09941794723272324, "rewards/total_composite/mean": 0.8106499910354614, "rewards/total_composite/std": 0.03432309255003929, "reward": 0.8106499910354614, "reward_std": 0.034323085099458694, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1474594920873642, "sampling/sampling_logp_difference/max": 1.389364242553711, "sampling/importance_sampling_ratio/min": 0.24923372268676758, "sampling/importance_sampling_ratio/mean": 1.0484529733657837, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.622205764055252, "clip_ratio/low_mean": 0.020454546436667442, "clip_ratio/low_min": 0.020454546436667442, "clip_ratio/high_mean": 0.10546988667920232, "clip_ratio/high_max": 0.10546988667920232, "clip_ratio/region_mean": 0.12592443311586976, "reward_total_mean": 0.8106499910354614, "reward_meter_mean": 0.992560625076294, "reward_meter_std": 0.00855978112667799, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9587268829345703, "reward_repeat_soft_std": 0.050369251519441605, "reward_judge_quality_mean": 0.39375001192092896, "reward_judge_quality_std": 0.09941794723272324, "reward_total_composite_mean": 0.8106499910354614, "reward_total_composite_std": 0.03432309255003929} {"timestamp_utc": "2026-04-12T23:14:45Z", "mode": "train", "global_step": 417, "epoch": 0.0418884982420894, "loss": 0.0847, "grad_norm": 8.001968383789062, "learning_rate": 8.73939393939394e-06, "num_tokens": 804098.0, "completions/mean_length": 98.75, "completions/min_length": 74.0, "completions/max_length": 119.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.75, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.9285807609558105, "rewards/meter/std": 0.13801667094230652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8394078016281128, "rewards/repeat_soft/std": 0.1369604617357254, "rewards/judge_quality/mean": 0.3137499988079071, "rewards/judge_quality/std": 0.19863374531269073, "rewards/total_composite/mean": 0.745927095413208, "rewards/total_composite/std": 0.09772931039333344, "reward": 0.745927095413208, "reward_std": 0.09772931039333344, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13536742329597473, "sampling/sampling_logp_difference/max": 1.730783462524414, "sampling/importance_sampling_ratio/min": 0.1771455556154251, "sampling/importance_sampling_ratio/mean": 1.0293668508529663, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3061815351247787, "clip_ratio/low_mean": 0.04256710642948747, "clip_ratio/low_min": 0.04256710642948747, "clip_ratio/high_mean": 0.06964209396392107, "clip_ratio/high_max": 0.06964209396392107, "clip_ratio/region_mean": 0.11220920039340854, "reward_total_mean": 0.745927095413208, "reward_meter_mean": 0.9285807609558105, "reward_meter_std": 0.13801667094230652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8394078016281128, "reward_repeat_soft_std": 0.1369604617357254, "reward_judge_quality_mean": 0.3137499988079071, "reward_judge_quality_std": 0.19863374531269073, "reward_total_composite_mean": 0.745927095413208, "reward_total_composite_std": 0.09772931039333344} {"timestamp_utc": "2026-04-12T23:14:52Z", "mode": "train", "global_step": 418, "epoch": 0.041988950276243095, "loss": 0.0457, "grad_norm": 15.859803199768066, "learning_rate": 8.736363636363638e-06, "num_tokens": 805803.0, "completions/mean_length": 53.125, "completions/min_length": 45.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.125, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.8196721076965332, "rewards/meter/std": 0.33972129225730896, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8878238201141357, "rewards/repeat_soft/std": 0.09231502562761307, "rewards/judge_quality/mean": 0.4437499940395355, "rewards/judge_quality/std": 0.2084595113992691, "rewards/total_composite/mean": 0.7407598495483398, "rewards/total_composite/std": 0.16826266050338745, "reward": 0.7407598495483398, "reward_std": 0.16826264560222626, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1389797478914261, "sampling/sampling_logp_difference/max": 1.6195526123046875, "sampling/importance_sampling_ratio/min": 0.19798725843429565, "sampling/importance_sampling_ratio/mean": 1.016070008277893, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9800810217857361, "clip_ratio/low_mean": 0.034908134723082185, "clip_ratio/low_min": 0.034908134723082185, "clip_ratio/high_mean": 0.08838540222495794, "clip_ratio/high_max": 0.08838540222495794, "clip_ratio/region_mean": 0.12329353694804013, "reward_total_mean": 0.7407598495483398, "reward_meter_mean": 0.8196721076965332, "reward_meter_std": 0.33972129225730896, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8878238201141357, "reward_repeat_soft_std": 0.09231502562761307, "reward_judge_quality_mean": 0.4437499940395355, "reward_judge_quality_std": 0.2084595113992691, "reward_total_composite_mean": 0.7407598495483398, "reward_total_composite_std": 0.16826266050338745} {"timestamp_utc": "2026-04-12T23:14:59Z", "mode": "train", "global_step": 419, "epoch": 0.04208940231039678, "loss": 0.0353, "grad_norm": 7.352740287780762, "learning_rate": 8.733333333333333e-06, "num_tokens": 808139.0, "completions/mean_length": 103.0, "completions/min_length": 88.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.0, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.925580620765686, "rewards/meter/std": 0.17920780181884766, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9962865114212036, "rewards/repeat_soft/std": 0.005543560720980167, "rewards/judge_quality/mean": 0.6612499952316284, "rewards/judge_quality/std": 0.2605728805065155, "rewards/total_composite/mean": 0.864514946937561, "rewards/total_composite/std": 0.09903687238693237, "reward": 0.864514946937561, "reward_std": 0.09903687983751297, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1288803219795227, "sampling/sampling_logp_difference/max": 1.2712068557739258, "sampling/importance_sampling_ratio/min": 0.280492901802063, "sampling/importance_sampling_ratio/mean": 1.0256997346878052, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2101072296500206, "clip_ratio/low_mean": 0.08157027512788773, "clip_ratio/low_min": 0.08157027512788773, "clip_ratio/high_mean": 0.02320244163274765, "clip_ratio/high_max": 0.02320244163274765, "clip_ratio/region_mean": 0.10477271676063538, "reward_total_mean": 0.864514946937561, "reward_meter_mean": 0.925580620765686, "reward_meter_std": 0.17920780181884766, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9962865114212036, "reward_repeat_soft_std": 0.005543560720980167, "reward_judge_quality_mean": 0.6612499952316284, "reward_judge_quality_std": 0.2605728805065155, "reward_total_composite_mean": 0.864514946937561, "reward_total_composite_std": 0.09903687238693237} {"timestamp_utc": "2026-04-12T23:15:06Z", "mode": "train", "global_step": 420, "epoch": 0.04218985434455048, "loss": 0.023, "grad_norm": 13.643331527709961, "learning_rate": 8.73030303030303e-06, "num_tokens": 809773.0, "completions/mean_length": 52.25, "completions/min_length": 44.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.25, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.5855252742767334, "rewards/meter/std": 0.44506746530532837, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9590322375297546, "rewards/repeat_soft/std": 0.09002047032117844, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.6335145831108093, "rewards/total_composite/std": 0.19549381732940674, "reward": 0.6335145831108093, "reward_std": 0.19549378752708435, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1878392994403839, "sampling/sampling_logp_difference/max": 1.8459452390670776, "sampling/importance_sampling_ratio/min": 0.15787601470947266, "sampling/importance_sampling_ratio/mean": 1.019885540008545, "sampling/importance_sampling_ratio/max": 1.9105006456375122, "entropy": 1.829648420214653, "clip_ratio/low_mean": 0.053845749236643314, "clip_ratio/low_min": 0.053845749236643314, "clip_ratio/high_mean": 0.11631151475012302, "clip_ratio/high_max": 0.11631151475012302, "clip_ratio/region_mean": 0.17015726398676634, "reward_total_mean": 0.6335145831108093, "reward_meter_mean": 0.5855252742767334, "reward_meter_std": 0.44506746530532837, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9590322375297546, "reward_repeat_soft_std": 0.09002047032117844, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.6335145831108093, "reward_total_composite_std": 0.19549381732940674} {"timestamp_utc": "2026-04-12T23:15:13Z", "mode": "train", "global_step": 421, "epoch": 0.04229030637870417, "loss": 0.0349, "grad_norm": 19.71310043334961, "learning_rate": 8.727272727272728e-06, "num_tokens": 811426.0, "completions/mean_length": 29.625, "completions/min_length": 24.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.625, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.7937754392623901, "rewards/meter/std": 0.31456121802330017, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9513249397277832, "rewards/repeat_soft/std": 0.02875583805143833, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7305814623832703, "rewards/total_composite/std": 0.14495150744915009, "reward": 0.7305814623832703, "reward_std": 0.1449514925479889, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1590035855770111, "sampling/sampling_logp_difference/max": 1.296311378479004, "sampling/importance_sampling_ratio/min": 0.2735389173030853, "sampling/importance_sampling_ratio/mean": 1.0613901615142822, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5776749104261398, "clip_ratio/low_mean": 0.07332717627286911, "clip_ratio/low_min": 0.07332717627286911, "clip_ratio/high_mean": 0.12988790310919285, "clip_ratio/high_max": 0.12988790310919285, "clip_ratio/region_mean": 0.20321507938206196, "reward_total_mean": 0.7305814623832703, "reward_meter_mean": 0.7937754392623901, "reward_meter_std": 0.31456121802330017, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9513249397277832, "reward_repeat_soft_std": 0.02875583805143833, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7305814623832703, "reward_total_composite_std": 0.14495150744915009} {"timestamp_utc": "2026-04-12T23:15:20Z", "mode": "train", "global_step": 422, "epoch": 0.04239075841285786, "loss": 0.0803, "grad_norm": 14.364946365356445, "learning_rate": 8.724242424242425e-06, "num_tokens": 813114.0, "completions/mean_length": 49.0, "completions/min_length": 33.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.714447021484375, "rewards/meter/std": 0.4050632119178772, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8922529220581055, "rewards/repeat_soft/std": 0.14730651676654816, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.21357084810733795, "rewards/total_composite/mean": 0.7103514671325684, "rewards/total_composite/std": 0.21625736355781555, "reward": 0.7103514671325684, "reward_std": 0.21625736355781555, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13579939305782318, "sampling/sampling_logp_difference/max": 1.0952553749084473, "sampling/importance_sampling_ratio/min": 0.33445417881011963, "sampling/importance_sampling_ratio/mean": 1.0035717487335205, "sampling/importance_sampling_ratio/max": 1.98849618434906, "entropy": 1.3966693207621574, "clip_ratio/low_mean": 0.04562068032100797, "clip_ratio/low_min": 0.04562068032100797, "clip_ratio/high_mean": 0.11016096360981464, "clip_ratio/high_max": 0.11016096360981464, "clip_ratio/region_mean": 0.1557816439308226, "reward_total_mean": 0.7103514671325684, "reward_meter_mean": 0.714447021484375, "reward_meter_std": 0.4050632119178772, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8922529220581055, "reward_repeat_soft_std": 0.14730651676654816, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.21357084810733795, "reward_total_composite_mean": 0.7103514671325684, "reward_total_composite_std": 0.21625736355781555} {"timestamp_utc": "2026-04-12T23:15:27Z", "mode": "train", "global_step": 423, "epoch": 0.042491210447011554, "loss": 0.05, "grad_norm": 10.484199523925781, "learning_rate": 8.72121212121212e-06, "num_tokens": 815224.0, "completions/mean_length": 90.75, "completions/min_length": 53.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.75, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.602098822593689, "rewards/meter/std": 0.33971330523490906, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9683234691619873, "rewards/repeat_soft/std": 0.039228785783052444, "rewards/judge_quality/mean": 0.45249998569488525, "rewards/judge_quality/std": 0.18100909888744354, "rewards/total_composite/mean": 0.6535268425941467, "rewards/total_composite/std": 0.17339988052845, "reward": 0.6535268425941467, "reward_std": 0.1733998954296112, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1978752166032791, "sampling/sampling_logp_difference/max": 1.4730768203735352, "sampling/importance_sampling_ratio/min": 0.22921913862228394, "sampling/importance_sampling_ratio/mean": 1.0324070453643799, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.065043106675148, "clip_ratio/low_mean": 0.07649808935821056, "clip_ratio/low_min": 0.07649808935821056, "clip_ratio/high_mean": 0.09552177414298058, "clip_ratio/high_max": 0.09552177414298058, "clip_ratio/region_mean": 0.17201986350119114, "reward_total_mean": 0.6535268425941467, "reward_meter_mean": 0.602098822593689, "reward_meter_std": 0.33971330523490906, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9683234691619873, "reward_repeat_soft_std": 0.039228785783052444, "reward_judge_quality_mean": 0.45249998569488525, "reward_judge_quality_std": 0.18100909888744354, "reward_total_composite_mean": 0.6535268425941467, "reward_total_composite_std": 0.17339988052845} {"timestamp_utc": "2026-04-12T23:15:33Z", "mode": "train", "global_step": 424, "epoch": 0.04259166248116524, "loss": 0.059, "grad_norm": 16.282878875732422, "learning_rate": 8.71818181818182e-06, "num_tokens": 817140.0, "completions/mean_length": 68.5, "completions/min_length": 51.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.5, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.8717867136001587, "rewards/meter/std": 0.19064652919769287, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9477397203445435, "rewards/repeat_soft/std": 0.10298575460910797, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.763077974319458, "rewards/total_composite/std": 0.08344349265098572, "reward": 0.763077974319458, "reward_std": 0.08344349265098572, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17412874102592468, "sampling/sampling_logp_difference/max": 1.1104965209960938, "sampling/importance_sampling_ratio/min": 0.3293953835964203, "sampling/importance_sampling_ratio/mean": 1.0266932249069214, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.776472732424736, "clip_ratio/low_mean": 0.06108739227056503, "clip_ratio/low_min": 0.06108739227056503, "clip_ratio/high_mean": 0.10932928742840886, "clip_ratio/high_max": 0.10932928742840886, "clip_ratio/region_mean": 0.1704166796989739, "reward_total_mean": 0.763077974319458, "reward_meter_mean": 0.8717867136001587, "reward_meter_std": 0.19064652919769287, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9477397203445435, "reward_repeat_soft_std": 0.10298575460910797, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.763077974319458, "reward_total_composite_std": 0.08344349265098572} {"timestamp_utc": "2026-04-12T23:15:40Z", "mode": "train", "global_step": 425, "epoch": 0.04269211451531894, "loss": 0.0601, "grad_norm": 27.168560028076172, "learning_rate": 8.715151515151515e-06, "num_tokens": 818805.0, "completions/mean_length": 27.125, "completions/min_length": 25.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.125, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.5827194452285767, "rewards/meter/std": 0.4273902177810669, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9874288439750671, "rewards/repeat_soft/std": 0.011401326395571232, "rewards/judge_quality/mean": 0.7699999809265137, "rewards/judge_quality/std": 0.22677870094776154, "rewards/total_composite/mean": 0.7419666051864624, "rewards/total_composite/std": 0.23377686738967896, "reward": 0.7419666051864624, "reward_std": 0.23377685248851776, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1188708022236824, "sampling/sampling_logp_difference/max": 1.2906403541564941, "sampling/importance_sampling_ratio/min": 0.27509456872940063, "sampling/importance_sampling_ratio/mean": 1.0000413656234741, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9232763797044754, "clip_ratio/low_mean": 0.05858262209221721, "clip_ratio/low_min": 0.05858262209221721, "clip_ratio/high_mean": 0.055839947424829006, "clip_ratio/high_max": 0.055839947424829006, "clip_ratio/region_mean": 0.11442256951704621, "reward_total_mean": 0.7419666051864624, "reward_meter_mean": 0.5827194452285767, "reward_meter_std": 0.4273902177810669, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9874288439750671, "reward_repeat_soft_std": 0.011401326395571232, "reward_judge_quality_mean": 0.7699999809265137, "reward_judge_quality_std": 0.22677870094776154, "reward_total_composite_mean": 0.7419666051864624, "reward_total_composite_std": 0.23377686738967896} {"timestamp_utc": "2026-04-12T23:15:46Z", "mode": "train", "global_step": 426, "epoch": 0.042792566549472624, "loss": -0.038, "grad_norm": 10.380043983459473, "learning_rate": 8.712121212121212e-06, "num_tokens": 820572.0, "completions/mean_length": 58.875, "completions/min_length": 52.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.875, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.7849419116973877, "rewards/meter/std": 0.3511310815811157, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9579784870147705, "rewards/repeat_soft/std": 0.03989151492714882, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.2515062093734741, "rewards/total_composite/mean": 0.6614232063293457, "rewards/total_composite/std": 0.32940465211868286, "reward": 0.6614232063293457, "reward_std": 0.32940465211868286, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20139729976654053, "sampling/sampling_logp_difference/max": 1.2309961318969727, "sampling/importance_sampling_ratio/min": 0.29200154542922974, "sampling/importance_sampling_ratio/mean": 1.0271530151367188, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7164409309625626, "clip_ratio/low_mean": 0.07467948831617832, "clip_ratio/low_min": 0.07467948831617832, "clip_ratio/high_mean": 0.11509432177990675, "clip_ratio/high_max": 0.11509432177990675, "clip_ratio/region_mean": 0.18977381009608507, "reward_total_mean": 0.6614232063293457, "reward_meter_mean": 0.7849419116973877, "reward_meter_std": 0.3511310815811157, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9579784870147705, "reward_repeat_soft_std": 0.03989151492714882, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.2515062093734741, "reward_total_composite_mean": 0.6614232063293457, "reward_total_composite_std": 0.32940465211868286} {"timestamp_utc": "2026-04-12T23:15:54Z", "mode": "train", "global_step": 427, "epoch": 0.04289301858362632, "loss": 0.0578, "grad_norm": 6.89548921585083, "learning_rate": 8.70909090909091e-06, "num_tokens": 823198.0, "completions/mean_length": 147.25, "completions/min_length": 119.0, "completions/max_length": 175.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 147.25, "completions/min_terminated_length": 119.0, "completions/max_terminated_length": 175.0, "rewards/meter/mean": 0.78412926197052, "rewards/meter/std": 0.2861679792404175, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9451888203620911, "rewards/repeat_soft/std": 0.12222348153591156, "rewards/judge_quality/mean": 0.5299999713897705, "rewards/judge_quality/std": 0.2285357266664505, "rewards/total_composite/mean": 0.7563770413398743, "rewards/total_composite/std": 0.1200062483549118, "reward": 0.7563770413398743, "reward_std": 0.1200062558054924, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1615327149629593, "sampling/sampling_logp_difference/max": 2.2313809394836426, "sampling/importance_sampling_ratio/min": 0.10738003998994827, "sampling/importance_sampling_ratio/mean": 1.0255837440490723, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5439204573631287, "clip_ratio/low_mean": 0.06102440319955349, "clip_ratio/low_min": 0.06102440319955349, "clip_ratio/high_mean": 0.07127992156893015, "clip_ratio/high_max": 0.07127992156893015, "clip_ratio/region_mean": 0.13230432476848364, "reward_total_mean": 0.7563770413398743, "reward_meter_mean": 0.78412926197052, "reward_meter_std": 0.2861679792404175, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9451888203620911, "reward_repeat_soft_std": 0.12222348153591156, "reward_judge_quality_mean": 0.5299999713897705, "reward_judge_quality_std": 0.2285357266664505, "reward_total_composite_mean": 0.7563770413398743, "reward_total_composite_std": 0.1200062483549118} {"timestamp_utc": "2026-04-12T23:16:01Z", "mode": "train", "global_step": 428, "epoch": 0.04299347061778001, "loss": 0.0304, "grad_norm": 15.515788078308105, "learning_rate": 8.706060606060607e-06, "num_tokens": 824870.0, "completions/mean_length": 42.0, "completions/min_length": 38.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.0, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.6622264385223389, "rewards/meter/std": 0.33998388051986694, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9989393949508667, "rewards/repeat_soft/std": 0.002044049324467778, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.6960208415985107, "rewards/total_composite/std": 0.1679197996854782, "reward": 0.6960208415985107, "reward_std": 0.1679197996854782, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1869424730539322, "sampling/sampling_logp_difference/max": 1.3337602615356445, "sampling/importance_sampling_ratio/min": 0.2634846270084381, "sampling/importance_sampling_ratio/mean": 1.0360627174377441, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8978291898965836, "clip_ratio/low_mean": 0.05240274593234062, "clip_ratio/low_min": 0.05240274593234062, "clip_ratio/high_mean": 0.11898304522037506, "clip_ratio/high_max": 0.11898304522037506, "clip_ratio/region_mean": 0.17138579115271568, "reward_total_mean": 0.6960208415985107, "reward_meter_mean": 0.6622264385223389, "reward_meter_std": 0.33998388051986694, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9989393949508667, "reward_repeat_soft_std": 0.002044049324467778, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.6960208415985107, "reward_total_composite_std": 0.1679197996854782} {"timestamp_utc": "2026-04-12T23:16:08Z", "mode": "train", "global_step": 429, "epoch": 0.0430939226519337, "loss": -0.0322, "grad_norm": 6.929233551025391, "learning_rate": 8.703030303030304e-06, "num_tokens": 827268.0, "completions/mean_length": 120.75, "completions/min_length": 105.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.75, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.9356905221939087, "rewards/meter/std": 0.13746945559978485, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9724094867706299, "rewards/repeat_soft/std": 0.02603420428931713, "rewards/judge_quality/mean": 0.39374998211860657, "rewards/judge_quality/std": 0.15638209879398346, "rewards/total_composite/mean": 0.684059739112854, "rewards/total_composite/std": 0.2901011109352112, "reward": 0.684059739112854, "reward_std": 0.2901011109352112, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19992434978485107, "sampling/sampling_logp_difference/max": 1.592648983001709, "sampling/importance_sampling_ratio/min": 0.2738623321056366, "sampling/importance_sampling_ratio/mean": 1.045350432395935, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.7989132404327393, "clip_ratio/low_mean": 0.03787594102323055, "clip_ratio/low_min": 0.03787594102323055, "clip_ratio/high_mean": 0.12172015383839607, "clip_ratio/high_max": 0.12172015383839607, "clip_ratio/region_mean": 0.15959609486162663, "reward_total_mean": 0.684059739112854, "reward_meter_mean": 0.9356905221939087, "reward_meter_std": 0.13746945559978485, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9724094867706299, "reward_repeat_soft_std": 0.02603420428931713, "reward_judge_quality_mean": 0.39374998211860657, "reward_judge_quality_std": 0.15638209879398346, "reward_total_composite_mean": 0.684059739112854, "reward_total_composite_std": 0.2901011109352112} {"timestamp_utc": "2026-04-12T23:16:14Z", "mode": "train", "global_step": 430, "epoch": 0.043194374686087396, "loss": -0.0978, "grad_norm": 17.669113159179688, "learning_rate": 8.700000000000001e-06, "num_tokens": 828508.0, "completions/mean_length": 24.0, "completions/min_length": 18.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.0, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.3892284333705902, "rewards/meter/std": 0.42710191011428833, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9600961208343506, "rewards/repeat_soft/std": 0.006799087394028902, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.6596624255180359, "rewards/total_composite/std": 0.17358513176441193, "reward": 0.6596624255180359, "reward_std": 0.17358513176441193, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1641015112400055, "sampling/sampling_logp_difference/max": 1.7448158264160156, "sampling/importance_sampling_ratio/min": 0.17467714846134186, "sampling/importance_sampling_ratio/mean": 1.0517590045928955, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5266574174165726, "clip_ratio/low_mean": 0.10830361396074295, "clip_ratio/low_min": 0.10830361396074295, "clip_ratio/high_mean": 0.05729564744979143, "clip_ratio/high_max": 0.05729564744979143, "clip_ratio/region_mean": 0.16559926141053438, "reward_total_mean": 0.6596624255180359, "reward_meter_mean": 0.3892284333705902, "reward_meter_std": 0.42710191011428833, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9600961208343506, "reward_repeat_soft_std": 0.006799087394028902, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.6596624255180359, "reward_total_composite_std": 0.17358513176441193} {"timestamp_utc": "2026-04-12T23:16:22Z", "mode": "train", "global_step": 431, "epoch": 0.043294826720241084, "loss": 0.0512, "grad_norm": 14.389640808105469, "learning_rate": 8.696969696969699e-06, "num_tokens": 830048.0, "completions/mean_length": 44.5, "completions/min_length": 37.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.5, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.3693164885044098, "rewards/meter/std": 0.3759056031703949, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9867936968803406, "rewards/repeat_soft/std": 0.022591859102249146, "rewards/judge_quality/mean": 0.6025000214576721, "rewards/judge_quality/std": 0.2516658902168274, "rewards/total_composite/mean": 0.5956217646598816, "rewards/total_composite/std": 0.1773366779088974, "reward": 0.5956217646598816, "reward_std": 0.1773366779088974, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17676353454589844, "sampling/sampling_logp_difference/max": 1.4544391632080078, "sampling/importance_sampling_ratio/min": 0.23353131115436554, "sampling/importance_sampling_ratio/mean": 1.0290045738220215, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5919075459241867, "clip_ratio/low_mean": 0.0754218976944685, "clip_ratio/low_min": 0.0754218976944685, "clip_ratio/high_mean": 0.07458209153264761, "clip_ratio/high_max": 0.07458209153264761, "clip_ratio/region_mean": 0.1500039892271161, "reward_total_mean": 0.5956217646598816, "reward_meter_mean": 0.3693164885044098, "reward_meter_std": 0.3759056031703949, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9867936968803406, "reward_repeat_soft_std": 0.022591859102249146, "reward_judge_quality_mean": 0.6025000214576721, "reward_judge_quality_std": 0.2516658902168274, "reward_total_composite_mean": 0.5956217646598816, "reward_total_composite_std": 0.1773366779088974} {"timestamp_utc": "2026-04-12T23:16:31Z", "mode": "train", "global_step": 432, "epoch": 0.04339527875439478, "loss": -0.0373, "grad_norm": 6.246907711029053, "learning_rate": 8.693939393939394e-06, "num_tokens": 832772.0, "completions/mean_length": 125.5, "completions/min_length": 105.0, "completions/max_length": 154.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.5, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 154.0, "rewards/meter/mean": 0.9680401086807251, "rewards/meter/std": 0.02021421678364277, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9815517067909241, "rewards/repeat_soft/std": 0.025748714804649353, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.6850097179412842, "rewards/total_composite/std": 0.2801055610179901, "reward": 0.6850097179412842, "reward_std": 0.2801055312156677, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19987118244171143, "sampling/sampling_logp_difference/max": 1.2343182563781738, "sampling/importance_sampling_ratio/min": 0.2910331189632416, "sampling/importance_sampling_ratio/mean": 1.0595637559890747, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.8805834501981735, "clip_ratio/low_mean": 0.02500000037252903, "clip_ratio/low_min": 0.02500000037252903, "clip_ratio/high_mean": 0.15010275319218636, "clip_ratio/high_max": 0.15010275319218636, "clip_ratio/region_mean": 0.17510275356471539, "reward_total_mean": 0.6850097179412842, "reward_meter_mean": 0.9680401086807251, "reward_meter_std": 0.02021421678364277, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9815517067909241, "reward_repeat_soft_std": 0.025748714804649353, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.6850097179412842, "reward_total_composite_std": 0.2801055610179901} {"timestamp_utc": "2026-04-12T23:16:43Z", "mode": "train", "global_step": 433, "epoch": 0.043495730788548466, "loss": -0.1302, "grad_norm": 2.165391206741333, "learning_rate": 8.690909090909091e-06, "num_tokens": 834512.0, "completions/mean_length": 110.5, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 53.142860412597656, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.8399171829223633, "rewards/meter/std": 0.30197471380233765, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9140486121177673, "rewards/repeat_soft/std": 0.1527128964662552, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.2594912648200989, "rewards/total_composite/mean": 0.6867904663085938, "rewards/total_composite/std": 0.29507118463516235, "reward": 0.6867904663085938, "reward_std": 0.29507118463516235, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15626339614391327, "sampling/sampling_logp_difference/max": 1.1653919219970703, "sampling/importance_sampling_ratio/min": 0.31180045008659363, "sampling/importance_sampling_ratio/mean": 1.0379847288131714, "sampling/importance_sampling_ratio/max": 1.9994099140167236, "entropy": 1.7401975318789482, "clip_ratio/low_mean": 0.014705882407724857, "clip_ratio/low_min": 0.014705882407724857, "clip_ratio/high_mean": 0.12285821419209242, "clip_ratio/high_max": 0.12285821419209242, "clip_ratio/region_mean": 0.13756409659981728, "reward_total_mean": 0.6867904663085938, "reward_meter_mean": 0.8399171829223633, "reward_meter_std": 0.30197471380233765, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9140486121177673, "reward_repeat_soft_std": 0.1527128964662552, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.2594912648200989, "reward_total_composite_mean": 0.6867904663085938, "reward_total_composite_std": 0.29507118463516235} {"timestamp_utc": "2026-04-12T23:16:49Z", "mode": "train", "global_step": 434, "epoch": 0.04359618282270216, "loss": 0.007, "grad_norm": 14.24094295501709, "learning_rate": 8.687878787878789e-06, "num_tokens": 836187.0, "completions/mean_length": 55.375, "completions/min_length": 45.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.375, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.7777681946754456, "rewards/meter/std": 0.2742709815502167, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9640134572982788, "rewards/repeat_soft/std": 0.05034282058477402, "rewards/judge_quality/mean": 0.6825000047683716, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.8011469841003418, "rewards/total_composite/std": 0.11899468302726746, "reward": 0.8011469841003418, "reward_std": 0.11899466812610626, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17065881192684174, "sampling/sampling_logp_difference/max": 1.377739429473877, "sampling/importance_sampling_ratio/min": 0.252147912979126, "sampling/importance_sampling_ratio/mean": 0.9940162301063538, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9139152243733406, "clip_ratio/low_mean": 0.05352947488427162, "clip_ratio/low_min": 0.05352947488427162, "clip_ratio/high_mean": 0.07624130509793758, "clip_ratio/high_max": 0.07624130509793758, "clip_ratio/region_mean": 0.1297707799822092, "reward_total_mean": 0.8011469841003418, "reward_meter_mean": 0.7777681946754456, "reward_meter_std": 0.2742709815502167, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9640134572982788, "reward_repeat_soft_std": 0.05034282058477402, "reward_judge_quality_mean": 0.6825000047683716, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.8011469841003418, "reward_total_composite_std": 0.11899468302726746} {"timestamp_utc": "2026-04-12T23:16:55Z", "mode": "train", "global_step": 435, "epoch": 0.04369663485685585, "loss": 0.0583, "grad_norm": 27.078083038330078, "learning_rate": 8.684848484848486e-06, "num_tokens": 837632.0, "completions/mean_length": 23.625, "completions/min_length": 17.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.625, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9066217541694641, "rewards/meter/std": 0.1738719493150711, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9548016786575317, "rewards/repeat_soft/std": 0.011878631077706814, "rewards/judge_quality/mean": 0.3387500047683716, "rewards/judge_quality/std": 0.09538455307483673, "rewards/total_composite/mean": 0.7550849914550781, "rewards/total_composite/std": 0.0720069408416748, "reward": 0.7550849914550781, "reward_std": 0.0720069482922554, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18114237487316132, "sampling/sampling_logp_difference/max": 1.2225031852722168, "sampling/importance_sampling_ratio/min": 0.29449209570884705, "sampling/importance_sampling_ratio/mean": 1.02423095703125, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.63126802444458, "clip_ratio/low_mean": 0.06321839243173599, "clip_ratio/low_min": 0.06321839243173599, "clip_ratio/high_mean": 0.13110610796138644, "clip_ratio/high_max": 0.13110610796138644, "clip_ratio/region_mean": 0.19432450039312243, "reward_total_mean": 0.7550849914550781, "reward_meter_mean": 0.9066217541694641, "reward_meter_std": 0.1738719493150711, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9548016786575317, "reward_repeat_soft_std": 0.011878631077706814, "reward_judge_quality_mean": 0.3387500047683716, "reward_judge_quality_std": 0.09538455307483673, "reward_total_composite_mean": 0.7550849914550781, "reward_total_composite_std": 0.0720069408416748} {"timestamp_utc": "2026-04-12T23:17:02Z", "mode": "train", "global_step": 436, "epoch": 0.04379708689100954, "loss": -0.0279, "grad_norm": 9.601164817810059, "learning_rate": 8.681818181818182e-06, "num_tokens": 839823.0, "completions/mean_length": 91.875, "completions/min_length": 76.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.875, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.9364590048789978, "rewards/meter/std": 0.042212143540382385, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9961820840835571, "rewards/repeat_soft/std": 0.002299386775121093, "rewards/judge_quality/mean": 0.4737499952316284, "rewards/judge_quality/std": 0.16291432082653046, "rewards/total_composite/mean": 0.8084622621536255, "rewards/total_composite/std": 0.06666874885559082, "reward": 0.8084622621536255, "reward_std": 0.06666873395442963, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17098484933376312, "sampling/sampling_logp_difference/max": 2.874861240386963, "sampling/importance_sampling_ratio/min": 0.05642396956682205, "sampling/importance_sampling_ratio/mean": 1.0396991968154907, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.784565344452858, "clip_ratio/low_mean": 0.08981337212026119, "clip_ratio/low_min": 0.08981337212026119, "clip_ratio/high_mean": 0.08913006074726582, "clip_ratio/high_max": 0.08913006074726582, "clip_ratio/region_mean": 0.178943432867527, "reward_total_mean": 0.8084622621536255, "reward_meter_mean": 0.9364590048789978, "reward_meter_std": 0.042212143540382385, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9961820840835571, "reward_repeat_soft_std": 0.002299386775121093, "reward_judge_quality_mean": 0.4737499952316284, "reward_judge_quality_std": 0.16291432082653046, "reward_total_composite_mean": 0.8084622621536255, "reward_total_composite_std": 0.06666874885559082} {"timestamp_utc": "2026-04-12T23:17:09Z", "mode": "train", "global_step": 437, "epoch": 0.04389753892516324, "loss": 0.0666, "grad_norm": 9.157220840454102, "learning_rate": 8.67878787878788e-06, "num_tokens": 842053.0, "completions/mean_length": 92.75, "completions/min_length": 81.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.75, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.9407258629798889, "rewards/meter/std": 0.10364113003015518, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9578771591186523, "rewards/repeat_soft/std": 0.08899172395467758, "rewards/judge_quality/mean": 0.42500001192092896, "rewards/judge_quality/std": 0.22551529109477997, "rewards/total_composite/mean": 0.7966143488883972, "rewards/total_composite/std": 0.08589332550764084, "reward": 0.7966143488883972, "reward_std": 0.08589331805706024, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15030913054943085, "sampling/sampling_logp_difference/max": 1.5390040874481201, "sampling/importance_sampling_ratio/min": 0.21459472179412842, "sampling/importance_sampling_ratio/mean": 1.0356183052062988, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5505438894033432, "clip_ratio/low_mean": 0.05696567287668586, "clip_ratio/low_min": 0.05696567287668586, "clip_ratio/high_mean": 0.06658268254250288, "clip_ratio/high_max": 0.06658268254250288, "clip_ratio/region_mean": 0.12354835541918874, "reward_total_mean": 0.7966143488883972, "reward_meter_mean": 0.9407258629798889, "reward_meter_std": 0.10364113003015518, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9578771591186523, "reward_repeat_soft_std": 0.08899172395467758, "reward_judge_quality_mean": 0.42500001192092896, "reward_judge_quality_std": 0.22551529109477997, "reward_total_composite_mean": 0.7966143488883972, "reward_total_composite_std": 0.08589332550764084} {"timestamp_utc": "2026-04-12T23:17:15Z", "mode": "train", "global_step": 438, "epoch": 0.043997990959316925, "loss": 0.0289, "grad_norm": 14.92218017578125, "learning_rate": 8.675757575757576e-06, "num_tokens": 843720.0, "completions/mean_length": 49.375, "completions/min_length": 34.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.375, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.7033980488777161, "rewards/meter/std": 0.4099227488040924, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9948642253875732, "rewards/repeat_soft/std": 0.006093386560678482, "rewards/judge_quality/mean": 0.5674999952316284, "rewards/judge_quality/std": 0.1916656494140625, "rewards/total_composite/mean": 0.7362655401229858, "rewards/total_composite/std": 0.18090790510177612, "reward": 0.7362655401229858, "reward_std": 0.18090789020061493, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1766577661037445, "sampling/sampling_logp_difference/max": 1.4099516868591309, "sampling/importance_sampling_ratio/min": 0.24415507912635803, "sampling/importance_sampling_ratio/mean": 1.0239568948745728, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9121709018945694, "clip_ratio/low_mean": 0.0454926323145628, "clip_ratio/low_min": 0.0454926323145628, "clip_ratio/high_mean": 0.1098181251436472, "clip_ratio/high_max": 0.1098181251436472, "clip_ratio/region_mean": 0.15531075745821, "reward_total_mean": 0.7362655401229858, "reward_meter_mean": 0.7033980488777161, "reward_meter_std": 0.4099227488040924, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9948642253875732, "reward_repeat_soft_std": 0.006093386560678482, "reward_judge_quality_mean": 0.5674999952316284, "reward_judge_quality_std": 0.1916656494140625, "reward_total_composite_mean": 0.7362655401229858, "reward_total_composite_std": 0.18090790510177612} {"timestamp_utc": "2026-04-12T23:17:22Z", "mode": "train", "global_step": 439, "epoch": 0.04409844299347062, "loss": 0.1667, "grad_norm": 12.60328197479248, "learning_rate": 8.672727272727273e-06, "num_tokens": 845342.0, "completions/mean_length": 50.75, "completions/min_length": 33.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.6114770174026489, "rewards/meter/std": 0.4112090766429901, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9910472631454468, "rewards/repeat_soft/std": 0.018118714913725853, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.6532694101333618, "rewards/total_composite/std": 0.2098148912191391, "reward": 0.6532694101333618, "reward_std": 0.2098148763179779, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19013889133930206, "sampling/sampling_logp_difference/max": 2.326509475708008, "sampling/importance_sampling_ratio/min": 0.09763594716787338, "sampling/importance_sampling_ratio/mean": 1.041027307510376, "sampling/importance_sampling_ratio/max": 1.866373062133789, "entropy": 2.328011319041252, "clip_ratio/low_mean": 0.04676048178225756, "clip_ratio/low_min": 0.04676048178225756, "clip_ratio/high_mean": 0.13594804145395756, "clip_ratio/high_max": 0.13594804145395756, "clip_ratio/region_mean": 0.18270852323621511, "reward_total_mean": 0.6532694101333618, "reward_meter_mean": 0.6114770174026489, "reward_meter_std": 0.4112090766429901, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9910472631454468, "reward_repeat_soft_std": 0.018118714913725853, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.6532694101333618, "reward_total_composite_std": 0.2098148912191391} {"timestamp_utc": "2026-04-12T23:17:30Z", "mode": "train", "global_step": 440, "epoch": 0.04419889502762431, "loss": 0.0196, "grad_norm": 9.668055534362793, "learning_rate": 8.66969696969697e-06, "num_tokens": 847986.0, "completions/mean_length": 117.5, "completions/min_length": 99.0, "completions/max_length": 163.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.5, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 163.0, "rewards/meter/mean": 0.8008937835693359, "rewards/meter/std": 0.26498210430145264, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9441074132919312, "rewards/repeat_soft/std": 0.11866484582424164, "rewards/judge_quality/mean": 0.3112500011920929, "rewards/judge_quality/std": 0.1860443502664566, "rewards/total_composite/mean": 0.6944379806518555, "rewards/total_composite/std": 0.14641381800174713, "reward": 0.6944379806518555, "reward_std": 0.14641381800174713, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1851826012134552, "sampling/sampling_logp_difference/max": 1.7322773933410645, "sampling/importance_sampling_ratio/min": 0.1768811196088791, "sampling/importance_sampling_ratio/mean": 1.0448429584503174, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.538330927491188, "clip_ratio/low_mean": 0.06125502195209265, "clip_ratio/low_min": 0.06125502195209265, "clip_ratio/high_mean": 0.10599872842431068, "clip_ratio/high_max": 0.10599872842431068, "clip_ratio/region_mean": 0.16725375037640333, "reward_total_mean": 0.6944379806518555, "reward_meter_mean": 0.8008937835693359, "reward_meter_std": 0.26498210430145264, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9441074132919312, "reward_repeat_soft_std": 0.11866484582424164, "reward_judge_quality_mean": 0.3112500011920929, "reward_judge_quality_std": 0.1860443502664566, "reward_total_composite_mean": 0.6944379806518555, "reward_total_composite_std": 0.14641381800174713} {"timestamp_utc": "2026-04-12T23:17:36Z", "mode": "train", "global_step": 441, "epoch": 0.044299347061778, "loss": 0.0517, "grad_norm": 17.479206085205078, "learning_rate": 8.666666666666668e-06, "num_tokens": 849352.0, "completions/mean_length": 31.75, "completions/min_length": 27.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.75, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.8566986322402954, "rewards/meter/std": 0.3264141082763672, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9863388538360596, "rewards/repeat_soft/std": 0.03441537171602249, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.853898286819458, "rewards/total_composite/std": 0.19867819547653198, "reward": 0.853898286819458, "reward_std": 0.19867821037769318, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1348220407962799, "sampling/sampling_logp_difference/max": 1.110522747039795, "sampling/importance_sampling_ratio/min": 0.400981605052948, "sampling/importance_sampling_ratio/mean": 1.0222198963165283, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0875507704913616, "clip_ratio/low_mean": 0.06783008761703968, "clip_ratio/low_min": 0.06783008761703968, "clip_ratio/high_mean": 0.07392149418592453, "clip_ratio/high_max": 0.07392149418592453, "clip_ratio/region_mean": 0.1417515818029642, "reward_total_mean": 0.853898286819458, "reward_meter_mean": 0.8566986322402954, "reward_meter_std": 0.3264141082763672, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9863388538360596, "reward_repeat_soft_std": 0.03441537171602249, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.853898286819458, "reward_total_composite_std": 0.19867819547653198} {"timestamp_utc": "2026-04-12T23:17:43Z", "mode": "train", "global_step": 442, "epoch": 0.04439979909593169, "loss": -0.0229, "grad_norm": 13.95291805267334, "learning_rate": 8.663636363636363e-06, "num_tokens": 850902.0, "completions/mean_length": 41.75, "completions/min_length": 32.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.75, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.5801884531974792, "rewards/meter/std": 0.4435848891735077, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9903296828269958, "rewards/repeat_soft/std": 0.01453430112451315, "rewards/judge_quality/mean": 0.4024999737739563, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.6308677792549133, "rewards/total_composite/std": 0.19595322012901306, "reward": 0.6308677792549133, "reward_std": 0.19595322012901306, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17702816426753998, "sampling/sampling_logp_difference/max": 1.57745361328125, "sampling/importance_sampling_ratio/min": 0.20650024712085724, "sampling/importance_sampling_ratio/mean": 1.0229361057281494, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9620563387870789, "clip_ratio/low_mean": 0.06630707532167435, "clip_ratio/low_min": 0.06630707532167435, "clip_ratio/high_mean": 0.08632143586874008, "clip_ratio/high_max": 0.08632143586874008, "clip_ratio/region_mean": 0.15262851119041443, "reward_total_mean": 0.6308677792549133, "reward_meter_mean": 0.5801884531974792, "reward_meter_std": 0.4435848891735077, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9903296828269958, "reward_repeat_soft_std": 0.01453430112451315, "reward_judge_quality_mean": 0.4024999737739563, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.6308677792549133, "reward_total_composite_std": 0.19595322012901306} {"timestamp_utc": "2026-04-12T23:17:50Z", "mode": "train", "global_step": 443, "epoch": 0.044500251130085385, "loss": -0.0108, "grad_norm": 11.943352699279785, "learning_rate": 8.660606060606062e-06, "num_tokens": 852881.0, "completions/mean_length": 71.375, "completions/min_length": 65.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.375, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.6063605546951294, "rewards/meter/std": 0.4112186133861542, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9723418951034546, "rewards/repeat_soft/std": 0.025245532393455505, "rewards/judge_quality/mean": 0.5137500166893005, "rewards/judge_quality/std": 0.1277204155921936, "rewards/total_composite/mean": 0.6679714918136597, "rewards/total_composite/std": 0.21762612462043762, "reward": 0.6679714918136597, "reward_std": 0.21762612462043762, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20011451840400696, "sampling/sampling_logp_difference/max": 1.220552921295166, "sampling/importance_sampling_ratio/min": 0.2950669825077057, "sampling/importance_sampling_ratio/mean": 1.0609068870544434, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.23347170650959, "clip_ratio/low_mean": 0.06602992676198483, "clip_ratio/low_min": 0.06602992676198483, "clip_ratio/high_mean": 0.10731012746691704, "clip_ratio/high_max": 0.10731012746691704, "clip_ratio/region_mean": 0.17334005422890186, "reward_total_mean": 0.6679714918136597, "reward_meter_mean": 0.6063605546951294, "reward_meter_std": 0.4112186133861542, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9723418951034546, "reward_repeat_soft_std": 0.025245532393455505, "reward_judge_quality_mean": 0.5137500166893005, "reward_judge_quality_std": 0.1277204155921936, "reward_total_composite_mean": 0.6679714918136597, "reward_total_composite_std": 0.21762612462043762} {"timestamp_utc": "2026-04-12T23:17:57Z", "mode": "train", "global_step": 444, "epoch": 0.04460070316423908, "loss": 0.1394, "grad_norm": 6.950952053070068, "learning_rate": 8.657575757575758e-06, "num_tokens": 855411.0, "completions/mean_length": 126.25, "completions/min_length": 91.0, "completions/max_length": 153.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.25, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 153.0, "rewards/meter/mean": 0.962444543838501, "rewards/meter/std": 0.07565572112798691, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.912335991859436, "rewards/repeat_soft/std": 0.10446178913116455, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.17266297340393066, "rewards/total_composite/mean": 0.7887086272239685, "rewards/total_composite/std": 0.06377964466810226, "reward": 0.7887086272239685, "reward_std": 0.06377964466810226, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15076377987861633, "sampling/sampling_logp_difference/max": 1.3234319686889648, "sampling/importance_sampling_ratio/min": 0.2662200927734375, "sampling/importance_sampling_ratio/mean": 1.0331894159317017, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.961112380027771, "clip_ratio/low_mean": 0.05008507426828146, "clip_ratio/low_min": 0.05008507426828146, "clip_ratio/high_mean": 0.08911538869142532, "clip_ratio/high_max": 0.08911538869142532, "clip_ratio/region_mean": 0.13920046295970678, "reward_total_mean": 0.7887086272239685, "reward_meter_mean": 0.962444543838501, "reward_meter_std": 0.07565572112798691, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.912335991859436, "reward_repeat_soft_std": 0.10446178913116455, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.17266297340393066, "reward_total_composite_mean": 0.7887086272239685, "reward_total_composite_std": 0.06377964466810226} {"timestamp_utc": "2026-04-12T23:18:05Z", "mode": "train", "global_step": 445, "epoch": 0.04470115519839277, "loss": 0.177, "grad_norm": 6.882579326629639, "learning_rate": 8.654545454545455e-06, "num_tokens": 858033.0, "completions/mean_length": 144.75, "completions/min_length": 95.0, "completions/max_length": 196.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 144.75, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 196.0, "rewards/meter/mean": 0.9780164361000061, "rewards/meter/std": 0.033467765897512436, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9325133562088013, "rewards/repeat_soft/std": 0.06715845316648483, "rewards/judge_quality/mean": 0.2549999952316284, "rewards/judge_quality/std": 0.111867256462574, "rewards/total_composite/mean": 0.7523586750030518, "rewards/total_composite/std": 0.04821430519223213, "reward": 0.7523586750030518, "reward_std": 0.048214323818683624, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16961638629436493, "sampling/sampling_logp_difference/max": 1.617447853088379, "sampling/importance_sampling_ratio/min": 0.19840441644191742, "sampling/importance_sampling_ratio/mean": 1.0525151491165161, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.212193012237549, "clip_ratio/low_mean": 0.05526641430333257, "clip_ratio/low_min": 0.05526641430333257, "clip_ratio/high_mean": 0.08498483151197433, "clip_ratio/high_max": 0.08498483151197433, "clip_ratio/region_mean": 0.1402512458153069, "reward_total_mean": 0.7523586750030518, "reward_meter_mean": 0.9780164361000061, "reward_meter_std": 0.033467765897512436, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9325133562088013, "reward_repeat_soft_std": 0.06715845316648483, "reward_judge_quality_mean": 0.2549999952316284, "reward_judge_quality_std": 0.111867256462574, "reward_total_composite_mean": 0.7523586750030518, "reward_total_composite_std": 0.04821430519223213} {"timestamp_utc": "2026-04-12T23:18:12Z", "mode": "train", "global_step": 446, "epoch": 0.04480160723254646, "loss": 0.1736, "grad_norm": 29.793655395507812, "learning_rate": 8.651515151515152e-06, "num_tokens": 859738.0, "completions/mean_length": 46.125, "completions/min_length": 28.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.125, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.7623186111450195, "rewards/meter/std": 0.3243028223514557, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9849733114242554, "rewards/repeat_soft/std": 0.02882581576704979, "rewards/judge_quality/mean": 0.5774999856948853, "rewards/judge_quality/std": 0.1609569787979126, "rewards/total_composite/mean": 0.758540689945221, "rewards/total_composite/std": 0.16675123572349548, "reward": 0.758540689945221, "reward_std": 0.16675123572349548, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14866042137145996, "sampling/sampling_logp_difference/max": 1.5851550102233887, "sampling/importance_sampling_ratio/min": 0.20491603016853333, "sampling/importance_sampling_ratio/mean": 0.9941482543945312, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9900865256786346, "clip_ratio/low_mean": 0.038825757801532745, "clip_ratio/low_min": 0.038825757801532745, "clip_ratio/high_mean": 0.15613035298883915, "clip_ratio/high_max": 0.15613035298883915, "clip_ratio/region_mean": 0.1949561107903719, "reward_total_mean": 0.758540689945221, "reward_meter_mean": 0.7623186111450195, "reward_meter_std": 0.3243028223514557, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9849733114242554, "reward_repeat_soft_std": 0.02882581576704979, "reward_judge_quality_mean": 0.5774999856948853, "reward_judge_quality_std": 0.1609569787979126, "reward_total_composite_mean": 0.758540689945221, "reward_total_composite_std": 0.16675123572349548} {"timestamp_utc": "2026-04-12T23:18:19Z", "mode": "train", "global_step": 447, "epoch": 0.04490205926670015, "loss": 0.0433, "grad_norm": 23.103769302368164, "learning_rate": 8.64848484848485e-06, "num_tokens": 861551.0, "completions/mean_length": 43.625, "completions/min_length": 40.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.625, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.8199383616447449, "rewards/meter/std": 0.2876770496368408, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9721415042877197, "rewards/repeat_soft/std": 0.041107550263404846, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.7631863951683044, "rewards/total_composite/std": 0.14952588081359863, "reward": 0.7631863951683044, "reward_std": 0.14952588081359863, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20923276245594025, "sampling/sampling_logp_difference/max": 1.5698919296264648, "sampling/importance_sampling_ratio/min": 0.2080676555633545, "sampling/importance_sampling_ratio/mean": 1.0403714179992676, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4300797134637833, "clip_ratio/low_mean": 0.043755664490163326, "clip_ratio/low_min": 0.043755664490163326, "clip_ratio/high_mean": 0.1135339867323637, "clip_ratio/high_max": 0.1135339867323637, "clip_ratio/region_mean": 0.15728965122252703, "reward_total_mean": 0.7631863951683044, "reward_meter_mean": 0.8199383616447449, "reward_meter_std": 0.2876770496368408, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9721415042877197, "reward_repeat_soft_std": 0.041107550263404846, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.7631863951683044, "reward_total_composite_std": 0.14952588081359863} {"timestamp_utc": "2026-04-12T23:18:26Z", "mode": "train", "global_step": 448, "epoch": 0.045002511300853844, "loss": 0.013, "grad_norm": 14.341288566589355, "learning_rate": 8.645454545454545e-06, "num_tokens": 863196.0, "completions/mean_length": 46.625, "completions/min_length": 41.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.625, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.8709509372711182, "rewards/meter/std": 0.19180485606193542, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9950623512268066, "rewards/repeat_soft/std": 0.005863826256245375, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.10392305999994278, "rewards/total_composite/mean": 0.7809341549873352, "rewards/total_composite/std": 0.09510692209005356, "reward": 0.7809341549873352, "reward_std": 0.09510691463947296, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21751703321933746, "sampling/sampling_logp_difference/max": 1.3758831024169922, "sampling/importance_sampling_ratio/min": 0.25261643528938293, "sampling/importance_sampling_ratio/mean": 1.0462380647659302, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.142319694161415, "clip_ratio/low_mean": 0.05329268239438534, "clip_ratio/low_min": 0.05329268239438534, "clip_ratio/high_mean": 0.10692180041223764, "clip_ratio/high_max": 0.10692180041223764, "clip_ratio/region_mean": 0.16021448280662298, "reward_total_mean": 0.7809341549873352, "reward_meter_mean": 0.8709509372711182, "reward_meter_std": 0.19180485606193542, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9950623512268066, "reward_repeat_soft_std": 0.005863826256245375, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.10392305999994278, "reward_total_composite_mean": 0.7809341549873352, "reward_total_composite_std": 0.09510692209005356} {"timestamp_utc": "2026-04-12T23:18:35Z", "mode": "train", "global_step": 449, "epoch": 0.04510296333500753, "loss": 0.1709, "grad_norm": 13.830315589904785, "learning_rate": 8.642424242424242e-06, "num_tokens": 864685.0, "completions/mean_length": 33.125, "completions/min_length": 24.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.125, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.8005282878875732, "rewards/meter/std": 0.2943017780780792, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9864086508750916, "rewards/repeat_soft/std": 0.027327554300427437, "rewards/judge_quality/mean": 0.6225000023841858, "rewards/judge_quality/std": 0.22403764724731445, "rewards/total_composite/mean": 0.7956286072731018, "rewards/total_composite/std": 0.18300634622573853, "reward": 0.7956286072731018, "reward_std": 0.18300634622573853, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10975746810436249, "sampling/sampling_logp_difference/max": 1.32759428024292, "sampling/importance_sampling_ratio/min": 0.35128653049468994, "sampling/importance_sampling_ratio/mean": 1.0361409187316895, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9810751117765903, "clip_ratio/low_mean": 0.05795684643089771, "clip_ratio/low_min": 0.05795684643089771, "clip_ratio/high_mean": 0.027370690368115902, "clip_ratio/high_max": 0.027370690368115902, "clip_ratio/region_mean": 0.08532753679901361, "reward_total_mean": 0.7956286072731018, "reward_meter_mean": 0.8005282878875732, "reward_meter_std": 0.2943017780780792, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9864086508750916, "reward_repeat_soft_std": 0.027327554300427437, "reward_judge_quality_mean": 0.6225000023841858, "reward_judge_quality_std": 0.22403764724731445, "reward_total_composite_mean": 0.7956286072731018, "reward_total_composite_std": 0.18300634622573853} {"timestamp_utc": "2026-04-12T23:18:41Z", "mode": "train", "global_step": 450, "epoch": 0.045203415369161226, "loss": 0.3352, "grad_norm": 18.755395889282227, "learning_rate": 8.63939393939394e-06, "num_tokens": 866262.0, "completions/mean_length": 47.125, "completions/min_length": 30.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.125, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.5960447788238525, "rewards/meter/std": 0.38606753945350647, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9969956278800964, "rewards/repeat_soft/std": 0.004343240987509489, "rewards/judge_quality/mean": 0.48374998569488525, "rewards/judge_quality/std": 0.18965664505958557, "rewards/total_composite/mean": 0.6127451062202454, "rewards/total_composite/std": 0.2980477809906006, "reward": 0.6127451062202454, "reward_std": 0.2980477511882782, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22084903717041016, "sampling/sampling_logp_difference/max": 1.394993782043457, "sampling/importance_sampling_ratio/min": 0.24783457815647125, "sampling/importance_sampling_ratio/mean": 1.0571175813674927, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.364145576953888, "clip_ratio/low_mean": 0.0650381064042449, "clip_ratio/low_min": 0.0650381064042449, "clip_ratio/high_mean": 0.14584106765687466, "clip_ratio/high_max": 0.14584106765687466, "clip_ratio/region_mean": 0.21087917406111956, "reward_total_mean": 0.6127451062202454, "reward_meter_mean": 0.5960447788238525, "reward_meter_std": 0.38606753945350647, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9969956278800964, "reward_repeat_soft_std": 0.004343240987509489, "reward_judge_quality_mean": 0.48374998569488525, "reward_judge_quality_std": 0.18965664505958557, "reward_total_composite_mean": 0.6127451062202454, "reward_total_composite_std": 0.2980477809906006} {"timestamp_utc": "2026-04-12T23:19:22Z", "mode": "eval", "global_step": 450, "epoch": 0.045203415369161226, "eval_loss": NaN, "eval_runtime": 41.277, "eval_samples_per_second": 1.938, "eval_steps_per_second": 0.242, "eval_num_tokens": 866262.0, "eval_completions/mean_length": 69.4625, "eval_completions/min_length": 27.6, "eval_completions/max_length": 121.2, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 69.4625, "eval_completions/min_terminated_length": 27.6, "eval_completions/max_terminated_length": 121.2, "eval_rewards/meter/mean": 0.668739840388298, "eval_rewards/meter/std": 0.3422308251261711, "eval_rewards/count_adherence/mean": 0.9637500047683716, "eval_rewards/count_adherence/std": 0.0818300984799862, "eval_rewards/hard_gate/mean": 0.925, "eval_rewards/hard_gate/std": 0.18771235942840575, "eval_rewards/repeat_soft/mean": 0.9885365962982178, "eval_rewards/repeat_soft/std": 0.01463707685470581, "eval_rewards/judge_quality/mean": 0.4292499929666519, "eval_rewards/judge_quality/std": 0.16611925289034843, "eval_rewards/total_composite/mean": 0.6271928191184998, "eval_rewards/total_composite/std": 0.2451991319656372, "eval_reward": 0.6271928191184998, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.156326724588871, "eval_sampling/sampling_logp_difference/max": 1.211293601989746, "eval_sampling/importance_sampling_ratio/min": 0.3090230628848076, "eval_sampling/importance_sampling_ratio/mean": 1.0501390099525452, "eval_sampling/importance_sampling_ratio/max": 1.590070903301239, "eval_entropy": 2.387557661533356, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6271928191184998, "eval_reward_meter_mean": 0.668739840388298, "eval_reward_meter_std": 0.3422308251261711, "eval_reward_count_adherence_mean": 0.9637500047683716, "eval_reward_count_adherence_std": 0.0818300984799862, "eval_reward_hard_gate_mean": 0.925, "eval_reward_hard_gate_std": 0.18771235942840575, "eval_reward_repeat_soft_mean": 0.9885365962982178, "eval_reward_repeat_soft_std": 0.01463707685470581, "eval_reward_judge_quality_mean": 0.4292499929666519, "eval_reward_judge_quality_std": 0.16611925289034843, "eval_reward_total_composite_mean": 0.6271928191184998, "eval_reward_total_composite_std": 0.2451991319656372} {"timestamp_utc": "2026-04-12T23:19:31Z", "mode": "train", "global_step": 451, "epoch": 0.045303867403314914, "loss": -0.008, "grad_norm": 9.851288795471191, "learning_rate": 8.636363636363637e-06, "num_tokens": 868298.0, "completions/mean_length": 72.5, "completions/min_length": 51.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.5, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.7815628051757812, "rewards/meter/std": 0.2986421585083008, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9938416481018066, "rewards/repeat_soft/std": 0.01006661169230938, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.14520922303199768, "rewards/total_composite/mean": 0.6360663771629333, "rewards/total_composite/std": 0.30158278346061707, "reward": 0.6360663771629333, "reward_std": 0.3015827536582947, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2158171832561493, "sampling/sampling_logp_difference/max": 1.2821907997131348, "sampling/importance_sampling_ratio/min": 0.27742883563041687, "sampling/importance_sampling_ratio/mean": 1.0316648483276367, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.871880292892456, "clip_ratio/low_mean": 0.03567538224160671, "clip_ratio/low_min": 0.03567538224160671, "clip_ratio/high_mean": 0.14200294762849808, "clip_ratio/high_max": 0.14200294762849808, "clip_ratio/region_mean": 0.1776783298701048, "reward_total_mean": 0.6360663771629333, "reward_meter_mean": 0.7815628051757812, "reward_meter_std": 0.2986421585083008, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9938416481018066, "reward_repeat_soft_std": 0.01006661169230938, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.14520922303199768, "reward_total_composite_mean": 0.6360663771629333, "reward_total_composite_std": 0.30158278346061707} {"timestamp_utc": "2026-04-12T23:19:38Z", "mode": "train", "global_step": 452, "epoch": 0.04540431943746861, "loss": 0.0629, "grad_norm": 8.94444465637207, "learning_rate": 8.633333333333334e-06, "num_tokens": 870941.0, "completions/mean_length": 113.375, "completions/min_length": 100.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.375, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.4145520329475403, "rewards/meter/std": 0.3624020218849182, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.08625821024179459, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9935540556907654, "rewards/repeat_soft/std": 0.006853767205029726, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5525288581848145, "rewards/total_composite/std": 0.16546082496643066, "reward": 0.5525288581848145, "reward_std": 0.16546082496643066, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2051757276058197, "sampling/sampling_logp_difference/max": 1.9577646255493164, "sampling/importance_sampling_ratio/min": 0.14117363095283508, "sampling/importance_sampling_ratio/mean": 1.047350287437439, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3810077011585236, "clip_ratio/low_mean": 0.09998427424579859, "clip_ratio/low_min": 0.09998427424579859, "clip_ratio/high_mean": 0.06250075623393059, "clip_ratio/high_max": 0.06250075623393059, "clip_ratio/region_mean": 0.16248503047972918, "reward_total_mean": 0.5525288581848145, "reward_meter_mean": 0.4145520329475403, "reward_meter_std": 0.3624020218849182, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.08625821024179459, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9935540556907654, "reward_repeat_soft_std": 0.006853767205029726, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5525288581848145, "reward_total_composite_std": 0.16546082496643066} {"timestamp_utc": "2026-04-12T23:19:46Z", "mode": "train", "global_step": 453, "epoch": 0.0455047714716223, "loss": -0.083, "grad_norm": 10.091301918029785, "learning_rate": 8.630303030303032e-06, "num_tokens": 872856.0, "completions/mean_length": 74.375, "completions/min_length": 43.0, "completions/max_length": 89.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.375, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 89.0, "rewards/meter/mean": 0.5161305069923401, "rewards/meter/std": 0.42014279961586, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.984504222869873, "rewards/repeat_soft/std": 0.02136831171810627, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.581334114074707, "rewards/total_composite/std": 0.19728514552116394, "reward": 0.581334114074707, "reward_std": 0.19728511571884155, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1909390240907669, "sampling/sampling_logp_difference/max": 1.3190784454345703, "sampling/importance_sampling_ratio/min": 0.26738157868385315, "sampling/importance_sampling_ratio/mean": 1.0252277851104736, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3356448113918304, "clip_ratio/low_mean": 0.09672854840755463, "clip_ratio/low_min": 0.09672854840755463, "clip_ratio/high_mean": 0.05483431462198496, "clip_ratio/high_max": 0.05483431462198496, "clip_ratio/region_mean": 0.15156286302953959, "reward_total_mean": 0.581334114074707, "reward_meter_mean": 0.5161305069923401, "reward_meter_std": 0.42014279961586, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.984504222869873, "reward_repeat_soft_std": 0.02136831171810627, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.581334114074707, "reward_total_composite_std": 0.19728514552116394} {"timestamp_utc": "2026-04-12T23:19:52Z", "mode": "train", "global_step": 454, "epoch": 0.04560522350577599, "loss": 0.0538, "grad_norm": 15.372694969177246, "learning_rate": 8.627272727272727e-06, "num_tokens": 874458.0, "completions/mean_length": 36.25, "completions/min_length": 32.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.25, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.8417208194732666, "rewards/meter/std": 0.23500235378742218, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9978598356246948, "rewards/repeat_soft/std": 0.003516496391966939, "rewards/judge_quality/mean": 0.4649999737739563, "rewards/judge_quality/std": 0.11563490331172943, "rewards/total_composite/mean": 0.7680603265762329, "rewards/total_composite/std": 0.12022081017494202, "reward": 0.7680603265762329, "reward_std": 0.12022079527378082, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18618905544281006, "sampling/sampling_logp_difference/max": 1.1652262210845947, "sampling/importance_sampling_ratio/min": 0.3118520975112915, "sampling/importance_sampling_ratio/mean": 1.051068663597107, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8358719199895859, "clip_ratio/low_mean": 0.023346656933426857, "clip_ratio/low_min": 0.023346656933426857, "clip_ratio/high_mean": 0.13759315432980657, "clip_ratio/high_max": 0.13759315432980657, "clip_ratio/region_mean": 0.16093981126323342, "reward_total_mean": 0.7680603265762329, "reward_meter_mean": 0.8417208194732666, "reward_meter_std": 0.23500235378742218, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9978598356246948, "reward_repeat_soft_std": 0.003516496391966939, "reward_judge_quality_mean": 0.4649999737739563, "reward_judge_quality_std": 0.11563490331172943, "reward_total_composite_mean": 0.7680603265762329, "reward_total_composite_std": 0.12022081017494202} {"timestamp_utc": "2026-04-12T23:19:59Z", "mode": "train", "global_step": 455, "epoch": 0.045705675539929685, "loss": 0.0697, "grad_norm": 8.666290283203125, "learning_rate": 8.624242424242424e-06, "num_tokens": 876747.0, "completions/mean_length": 102.125, "completions/min_length": 79.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.125, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.914614737033844, "rewards/meter/std": 0.1331835389137268, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9878299236297607, "rewards/repeat_soft/std": 0.013982153497636318, "rewards/judge_quality/mean": 0.3512499928474426, "rewards/judge_quality/std": 0.252045214176178, "rewards/total_composite/mean": 0.7610471248626709, "rewards/total_composite/std": 0.10284329950809479, "reward": 0.7610471248626709, "reward_std": 0.10284329950809479, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1906580626964569, "sampling/sampling_logp_difference/max": 1.4877400398254395, "sampling/importance_sampling_ratio/min": 0.3154601752758026, "sampling/importance_sampling_ratio/mean": 1.0419601202011108, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.892320930957794, "clip_ratio/low_mean": 0.07168858963996172, "clip_ratio/low_min": 0.07168858963996172, "clip_ratio/high_mean": 0.08013962768018246, "clip_ratio/high_max": 0.08013962768018246, "clip_ratio/region_mean": 0.15182821732014418, "reward_total_mean": 0.7610471248626709, "reward_meter_mean": 0.914614737033844, "reward_meter_std": 0.1331835389137268, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9878299236297607, "reward_repeat_soft_std": 0.013982153497636318, "reward_judge_quality_mean": 0.3512499928474426, "reward_judge_quality_std": 0.252045214176178, "reward_total_composite_mean": 0.7610471248626709, "reward_total_composite_std": 0.10284329950809479} {"timestamp_utc": "2026-04-12T23:20:06Z", "mode": "train", "global_step": 456, "epoch": 0.04580612757408337, "loss": 0.0252, "grad_norm": 19.901609420776367, "learning_rate": 8.621212121212122e-06, "num_tokens": 878441.0, "completions/mean_length": 37.75, "completions/min_length": 28.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.75, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.43395674228668213, "rewards/meter/std": 0.39264431595802307, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9996860027313232, "rewards/repeat_soft/std": 0.0008881183457560837, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.5329451560974121, "rewards/total_composite/std": 0.3491443991661072, "reward": 0.5329451560974121, "reward_std": 0.3491443991661072, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18590229749679565, "sampling/sampling_logp_difference/max": 1.4785947799682617, "sampling/importance_sampling_ratio/min": 0.2279578149318695, "sampling/importance_sampling_ratio/mean": 1.0035794973373413, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3422477021813393, "clip_ratio/low_mean": 0.05425307247787714, "clip_ratio/low_min": 0.05425307247787714, "clip_ratio/high_mean": 0.11646304838359356, "clip_ratio/high_max": 0.11646304838359356, "clip_ratio/region_mean": 0.1707161208614707, "reward_total_mean": 0.5329451560974121, "reward_meter_mean": 0.43395674228668213, "reward_meter_std": 0.39264431595802307, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9996860027313232, "reward_repeat_soft_std": 0.0008881183457560837, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.5329451560974121, "reward_total_composite_std": 0.3491443991661072} {"timestamp_utc": "2026-04-12T23:20:12Z", "mode": "train", "global_step": 457, "epoch": 0.04590657960823707, "loss": 0.0277, "grad_norm": 17.226221084594727, "learning_rate": 8.618181818181819e-06, "num_tokens": 879914.0, "completions/mean_length": 40.125, "completions/min_length": 31.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.125, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.8449429273605347, "rewards/meter/std": 0.21029865741729736, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9617567658424377, "rewards/repeat_soft/std": 0.044282421469688416, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.8273999691009521, "rewards/total_composite/std": 0.07279232889413834, "reward": 0.8273999691009521, "reward_std": 0.07279232889413834, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18386690318584442, "sampling/sampling_logp_difference/max": 1.3059649467468262, "sampling/importance_sampling_ratio/min": 0.27091097831726074, "sampling/importance_sampling_ratio/mean": 1.030527949333191, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9034367948770523, "clip_ratio/low_mean": 0.09608208853751421, "clip_ratio/low_min": 0.09608208853751421, "clip_ratio/high_mean": 0.03887043334543705, "clip_ratio/high_max": 0.03887043334543705, "clip_ratio/region_mean": 0.13495252188295126, "reward_total_mean": 0.8273999691009521, "reward_meter_mean": 0.8449429273605347, "reward_meter_std": 0.21029865741729736, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9617567658424377, "reward_repeat_soft_std": 0.044282421469688416, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.8273999691009521, "reward_total_composite_std": 0.07279232889413834} {"timestamp_utc": "2026-04-12T23:20:18Z", "mode": "train", "global_step": 458, "epoch": 0.046007031642390755, "loss": -0.0066, "grad_norm": 20.90218734741211, "learning_rate": 8.615151515151516e-06, "num_tokens": 881340.0, "completions/mean_length": 21.25, "completions/min_length": 19.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.25, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.5560683012008667, "rewards/meter/std": 0.43911802768707275, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9584887027740479, "rewards/repeat_soft/std": 0.008763790130615234, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.19334925711154938, "rewards/total_composite/mean": 0.6367045640945435, "rewards/total_composite/std": 0.17790387570858002, "reward": 0.6367045640945435, "reward_std": 0.17790387570858002, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2008567899465561, "sampling/sampling_logp_difference/max": 1.6547985076904297, "sampling/importance_sampling_ratio/min": 0.19113056361675262, "sampling/importance_sampling_ratio/mean": 1.0498888492584229, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0506961941719055, "clip_ratio/low_mean": 0.058941511902958155, "clip_ratio/low_min": 0.058941511902958155, "clip_ratio/high_mean": 0.08717461302876472, "clip_ratio/high_max": 0.08717461302876472, "clip_ratio/region_mean": 0.14611612493172288, "reward_total_mean": 0.6367045640945435, "reward_meter_mean": 0.5560683012008667, "reward_meter_std": 0.43911802768707275, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9584887027740479, "reward_repeat_soft_std": 0.008763790130615234, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.19334925711154938, "reward_total_composite_mean": 0.6367045640945435, "reward_total_composite_std": 0.17790387570858002} {"timestamp_utc": "2026-04-12T23:20:25Z", "mode": "train", "global_step": 459, "epoch": 0.04610748367654445, "loss": 0.1252, "grad_norm": 20.79795265197754, "learning_rate": 8.612121212121213e-06, "num_tokens": 883010.0, "completions/mean_length": 45.75, "completions/min_length": 36.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.75, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.7238732576370239, "rewards/meter/std": 0.4231775999069214, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9966292381286621, "rewards/repeat_soft/std": 0.007832803763449192, "rewards/judge_quality/mean": 0.5800000429153442, "rewards/judge_quality/std": 0.1884523630142212, "rewards/total_composite/mean": 0.7494058609008789, "rewards/total_composite/std": 0.21575051546096802, "reward": 0.7494058609008789, "reward_std": 0.21575051546096802, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21371857821941376, "sampling/sampling_logp_difference/max": 1.9703669548034668, "sampling/importance_sampling_ratio/min": 0.13940568268299103, "sampling/importance_sampling_ratio/mean": 1.0378260612487793, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.222277209162712, "clip_ratio/low_mean": 0.031901041977107525, "clip_ratio/low_min": 0.031901041977107525, "clip_ratio/high_mean": 0.1439657211303711, "clip_ratio/high_max": 0.1439657211303711, "clip_ratio/region_mean": 0.17586676310747862, "reward_total_mean": 0.7494058609008789, "reward_meter_mean": 0.7238732576370239, "reward_meter_std": 0.4231775999069214, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9966292381286621, "reward_repeat_soft_std": 0.007832803763449192, "reward_judge_quality_mean": 0.5800000429153442, "reward_judge_quality_std": 0.1884523630142212, "reward_total_composite_mean": 0.7494058609008789, "reward_total_composite_std": 0.21575051546096802} {"timestamp_utc": "2026-04-12T23:20:31Z", "mode": "train", "global_step": 460, "epoch": 0.046207935710698145, "loss": 0.1063, "grad_norm": 13.587638854980469, "learning_rate": 8.60909090909091e-06, "num_tokens": 885086.0, "completions/mean_length": 75.5, "completions/min_length": 50.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.5, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.3803871273994446, "rewards/meter/std": 0.25494661927223206, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9937744140625, "rewards/repeat_soft/std": 0.004149881657212973, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.5236766338348389, "rewards/total_composite/std": 0.1302151381969452, "reward": 0.5236766338348389, "reward_std": 0.130215123295784, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19353869557380676, "sampling/sampling_logp_difference/max": 1.899148941040039, "sampling/importance_sampling_ratio/min": 0.1496959626674652, "sampling/importance_sampling_ratio/mean": 1.0103038549423218, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6577086299657822, "clip_ratio/low_mean": 0.08220901526510715, "clip_ratio/low_min": 0.08220901526510715, "clip_ratio/high_mean": 0.09787844587117434, "clip_ratio/high_max": 0.09787844587117434, "clip_ratio/region_mean": 0.1800874611362815, "reward_total_mean": 0.5236766338348389, "reward_meter_mean": 0.3803871273994446, "reward_meter_std": 0.25494661927223206, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9937744140625, "reward_repeat_soft_std": 0.004149881657212973, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.5236766338348389, "reward_total_composite_std": 0.1302151381969452} {"timestamp_utc": "2026-04-12T23:20:37Z", "mode": "train", "global_step": 461, "epoch": 0.04630838774485183, "loss": 0.0568, "grad_norm": 21.839380264282227, "learning_rate": 8.606060606060606e-06, "num_tokens": 886635.0, "completions/mean_length": 29.625, "completions/min_length": 25.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.625, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.7121713161468506, "rewards/meter/std": 0.2743678390979767, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9955670237541199, "rewards/repeat_soft/std": 0.008184845559298992, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.21084441244602203, "rewards/total_composite/mean": 0.7219088077545166, "rewards/total_composite/std": 0.15774929523468018, "reward": 0.7219088077545166, "reward_std": 0.15774931013584137, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22285087406635284, "sampling/sampling_logp_difference/max": 1.9360859394073486, "sampling/importance_sampling_ratio/min": 0.1442675143480301, "sampling/importance_sampling_ratio/mean": 1.0138145685195923, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7142591923475266, "clip_ratio/low_mean": 0.06294998340308666, "clip_ratio/low_min": 0.06294998340308666, "clip_ratio/high_mean": 0.1281342599540949, "clip_ratio/high_max": 0.1281342599540949, "clip_ratio/region_mean": 0.19108424335718155, "reward_total_mean": 0.7219088077545166, "reward_meter_mean": 0.7121713161468506, "reward_meter_std": 0.2743678390979767, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9955670237541199, "reward_repeat_soft_std": 0.008184845559298992, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.21084441244602203, "reward_total_composite_mean": 0.7219088077545166, "reward_total_composite_std": 0.15774929523468018} {"timestamp_utc": "2026-04-12T23:20:44Z", "mode": "train", "global_step": 462, "epoch": 0.04640883977900553, "loss": 0.0481, "grad_norm": 13.02568244934082, "learning_rate": 8.603030303030303e-06, "num_tokens": 888433.0, "completions/mean_length": 61.75, "completions/min_length": 52.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.75, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.8600386381149292, "rewards/meter/std": 0.29712286591529846, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9993306398391724, "rewards/repeat_soft/std": 0.0018932786770164967, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.7678254246711731, "rewards/total_composite/std": 0.15833136439323425, "reward": 0.7678254246711731, "reward_std": 0.15833136439323425, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1623956859111786, "sampling/sampling_logp_difference/max": 1.5527958869934082, "sampling/importance_sampling_ratio/min": 0.2116553783416748, "sampling/importance_sampling_ratio/mean": 1.033503532409668, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8564894497394562, "clip_ratio/low_mean": 0.01607142947614193, "clip_ratio/low_min": 0.01607142947614193, "clip_ratio/high_mean": 0.13239080645143986, "clip_ratio/high_max": 0.13239080645143986, "clip_ratio/region_mean": 0.1484622359275818, "reward_total_mean": 0.7678254246711731, "reward_meter_mean": 0.8600386381149292, "reward_meter_std": 0.29712286591529846, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9993306398391724, "reward_repeat_soft_std": 0.0018932786770164967, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.7678254246711731, "reward_total_composite_std": 0.15833136439323425} {"timestamp_utc": "2026-04-12T23:20:51Z", "mode": "train", "global_step": 463, "epoch": 0.046509291813159215, "loss": 0.0675, "grad_norm": 13.812843322753906, "learning_rate": 8.6e-06, "num_tokens": 889992.0, "completions/mean_length": 41.875, "completions/min_length": 34.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.875, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.4513362944126129, "rewards/meter/std": 0.298054575920105, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9899604916572571, "rewards/repeat_soft/std": 0.013202440924942493, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.60622239112854, "rewards/total_composite/std": 0.11469129472970963, "reward": 0.60622239112854, "reward_std": 0.11469127982854843, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.247202068567276, "sampling/sampling_logp_difference/max": 4.244635581970215, "sampling/importance_sampling_ratio/min": 0.01434095948934555, "sampling/importance_sampling_ratio/mean": 1.0382658243179321, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.24991013109684, "clip_ratio/low_mean": 0.12404216127470136, "clip_ratio/low_min": 0.12404216127470136, "clip_ratio/high_mean": 0.0505640241317451, "clip_ratio/high_max": 0.0505640241317451, "clip_ratio/region_mean": 0.17460618540644646, "reward_total_mean": 0.60622239112854, "reward_meter_mean": 0.4513362944126129, "reward_meter_std": 0.298054575920105, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9899604916572571, "reward_repeat_soft_std": 0.013202440924942493, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.60622239112854, "reward_total_composite_std": 0.11469129472970963} {"timestamp_utc": "2026-04-12T23:20:57Z", "mode": "train", "global_step": 464, "epoch": 0.04660974384731291, "loss": -0.0529, "grad_norm": 13.543258666992188, "learning_rate": 8.596969696969698e-06, "num_tokens": 891571.0, "completions/mean_length": 33.375, "completions/min_length": 30.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.375, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.42223066091537476, "rewards/meter/std": 0.4493068754673004, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9762974977493286, "rewards/repeat_soft/std": 0.014409983530640602, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.5835085511207581, "rewards/total_composite/std": 0.19279791414737701, "reward": 0.5835085511207581, "reward_std": 0.19279791414737701, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12016096711158752, "sampling/sampling_logp_difference/max": 1.1314496994018555, "sampling/importance_sampling_ratio/min": 0.3225652873516083, "sampling/importance_sampling_ratio/mean": 1.024914026260376, "sampling/importance_sampling_ratio/max": 1.9165679216384888, "entropy": 1.1575526669621468, "clip_ratio/low_mean": 0.06444892473518848, "clip_ratio/low_min": 0.06444892473518848, "clip_ratio/high_mean": 0.0484571629203856, "clip_ratio/high_max": 0.0484571629203856, "clip_ratio/region_mean": 0.11290608765557408, "reward_total_mean": 0.5835085511207581, "reward_meter_mean": 0.42223066091537476, "reward_meter_std": 0.4493068754673004, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9762974977493286, "reward_repeat_soft_std": 0.014409983530640602, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.5835085511207581, "reward_total_composite_std": 0.19279791414737701} {"timestamp_utc": "2026-04-12T23:21:03Z", "mode": "train", "global_step": 465, "epoch": 0.0467101958814666, "loss": 0.0886, "grad_norm": 21.937891006469727, "learning_rate": 8.593939393939395e-06, "num_tokens": 892847.0, "completions/mean_length": 22.5, "completions/min_length": 19.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.5, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.8658735752105713, "rewards/meter/std": 0.24087516963481903, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.7566431760787964, "rewards/total_composite/std": 0.10618992149829865, "reward": 0.7566431760787964, "reward_std": 0.10618993639945984, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1905650943517685, "sampling/sampling_logp_difference/max": 1.1821870803833008, "sampling/importance_sampling_ratio/min": 0.3066074252128601, "sampling/importance_sampling_ratio/mean": 1.0348246097564697, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.254527598619461, "clip_ratio/low_mean": 0.02083333395421505, "clip_ratio/low_min": 0.02083333395421505, "clip_ratio/high_mean": 0.19369245041161776, "clip_ratio/high_max": 0.19369245041161776, "clip_ratio/region_mean": 0.2145257843658328, "reward_total_mean": 0.7566431760787964, "reward_meter_mean": 0.8658735752105713, "reward_meter_std": 0.24087516963481903, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.7566431760787964, "reward_total_composite_std": 0.10618992149829865} {"timestamp_utc": "2026-04-12T23:21:09Z", "mode": "train", "global_step": 466, "epoch": 0.04681064791562029, "loss": 0.0574, "grad_norm": 11.71673583984375, "learning_rate": 8.590909090909092e-06, "num_tokens": 894730.0, "completions/mean_length": 70.375, "completions/min_length": 63.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.375, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.5999476909637451, "rewards/meter/std": 0.4096795916557312, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9850830435752869, "rewards/repeat_soft/std": 0.015505336225032806, "rewards/judge_quality/mean": 0.47749999165534973, "rewards/judge_quality/std": 0.22211645543575287, "rewards/total_composite/mean": 0.6078383922576904, "rewards/total_composite/std": 0.31052854657173157, "reward": 0.6078383922576904, "reward_std": 0.31052854657173157, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21823406219482422, "sampling/sampling_logp_difference/max": 1.8511466979980469, "sampling/importance_sampling_ratio/min": 0.15705697238445282, "sampling/importance_sampling_ratio/mean": 1.0551923513412476, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.9800692349672318, "clip_ratio/low_mean": 0.054820626974105835, "clip_ratio/low_min": 0.054820626974105835, "clip_ratio/high_mean": 0.11422239895910025, "clip_ratio/high_max": 0.11422239895910025, "clip_ratio/region_mean": 0.16904302593320608, "reward_total_mean": 0.6078383922576904, "reward_meter_mean": 0.5999476909637451, "reward_meter_std": 0.4096795916557312, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9850830435752869, "reward_repeat_soft_std": 0.015505336225032806, "reward_judge_quality_mean": 0.47749999165534973, "reward_judge_quality_std": 0.22211645543575287, "reward_total_composite_mean": 0.6078383922576904, "reward_total_composite_std": 0.31052854657173157} {"timestamp_utc": "2026-04-12T23:21:15Z", "mode": "train", "global_step": 467, "epoch": 0.04691109994977398, "loss": 0.0746, "grad_norm": 27.435218811035156, "learning_rate": 8.587878787878788e-06, "num_tokens": 896290.0, "completions/mean_length": 30.0, "completions/min_length": 22.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.0, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.6044333577156067, "rewards/meter/std": 0.3662070333957672, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9862504005432129, "rewards/repeat_soft/std": 0.018274299800395966, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5461598634719849, "rewards/total_composite/std": 0.26763853430747986, "reward": 0.5461598634719849, "reward_std": 0.26763850450515747, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20848587155342102, "sampling/sampling_logp_difference/max": 1.178513526916504, "sampling/importance_sampling_ratio/min": 0.3077358305454254, "sampling/importance_sampling_ratio/mean": 1.0399103164672852, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8842953741550446, "clip_ratio/low_mean": 0.10127960331737995, "clip_ratio/low_min": 0.10127960331737995, "clip_ratio/high_mean": 0.07575757801532745, "clip_ratio/high_max": 0.07575757801532745, "clip_ratio/region_mean": 0.1770371813327074, "reward_total_mean": 0.5461598634719849, "reward_meter_mean": 0.6044333577156067, "reward_meter_std": 0.3662070333957672, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9862504005432129, "reward_repeat_soft_std": 0.018274299800395966, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5461598634719849, "reward_total_composite_std": 0.26763853430747986} {"timestamp_utc": "2026-04-12T23:21:27Z", "mode": "train", "global_step": 468, "epoch": 0.047011551983927674, "loss": -0.1433, "grad_norm": 2.8854830265045166, "learning_rate": 8.584848484848485e-06, "num_tokens": 898074.0, "completions/mean_length": 112.0, "completions/min_length": 43.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 54.857147216796875, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.5678571462631226, "rewards/meter/std": 0.2775091528892517, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9923787117004395, "rewards/repeat_soft/std": 0.007177090272307396, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.5782654285430908, "rewards/total_composite/std": 0.2508188486099243, "reward": 0.5782654285430908, "reward_std": 0.2508188486099243, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21288028359413147, "sampling/sampling_logp_difference/max": 2.30533504486084, "sampling/importance_sampling_ratio/min": 0.0997253879904747, "sampling/importance_sampling_ratio/mean": 1.0140104293823242, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.547509714961052, "clip_ratio/low_mean": 0.009433962404727936, "clip_ratio/low_min": 0.009433962404727936, "clip_ratio/high_mean": 0.1728997752070427, "clip_ratio/high_max": 0.1728997752070427, "clip_ratio/region_mean": 0.18233373761177063, "reward_total_mean": 0.5782654285430908, "reward_meter_mean": 0.5678571462631226, "reward_meter_std": 0.2775091528892517, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9923787117004395, "reward_repeat_soft_std": 0.007177090272307396, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.5782654285430908, "reward_total_composite_std": 0.2508188486099243} {"timestamp_utc": "2026-04-12T23:21:34Z", "mode": "train", "global_step": 469, "epoch": 0.04711200401808137, "loss": 0.1395, "grad_norm": 13.389942169189453, "learning_rate": 8.581818181818183e-06, "num_tokens": 899696.0, "completions/mean_length": 46.75, "completions/min_length": 39.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.75, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.7282257080078125, "rewards/meter/std": 0.3581485450267792, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9955543279647827, "rewards/repeat_soft/std": 0.009121197275817394, "rewards/judge_quality/mean": 0.41874998807907104, "rewards/judge_quality/std": 0.14574317634105682, "rewards/total_composite/mean": 0.6935069561004639, "rewards/total_composite/std": 0.20084483921527863, "reward": 0.6935069561004639, "reward_std": 0.20084482431411743, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21318060159683228, "sampling/sampling_logp_difference/max": 1.1256113052368164, "sampling/importance_sampling_ratio/min": 0.32445406913757324, "sampling/importance_sampling_ratio/mean": 1.0444365739822388, "sampling/importance_sampling_ratio/max": 1.9020357131958008, "entropy": 3.0094098448753357, "clip_ratio/low_mean": 0.03307109698653221, "clip_ratio/low_min": 0.03307109698653221, "clip_ratio/high_mean": 0.12092806585133076, "clip_ratio/high_max": 0.12092806585133076, "clip_ratio/region_mean": 0.15399916283786297, "reward_total_mean": 0.6935069561004639, "reward_meter_mean": 0.7282257080078125, "reward_meter_std": 0.3581485450267792, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9955543279647827, "reward_repeat_soft_std": 0.009121197275817394, "reward_judge_quality_mean": 0.41874998807907104, "reward_judge_quality_std": 0.14574317634105682, "reward_total_composite_mean": 0.6935069561004639, "reward_total_composite_std": 0.20084483921527863} {"timestamp_utc": "2026-04-12T23:21:41Z", "mode": "train", "global_step": 470, "epoch": 0.047212456052235056, "loss": 0.0153, "grad_norm": 15.639862060546875, "learning_rate": 8.57878787878788e-06, "num_tokens": 901333.0, "completions/mean_length": 39.625, "completions/min_length": 29.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.625, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.8690699934959412, "rewards/meter/std": 0.1305169016122818, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9896947145462036, "rewards/repeat_soft/std": 0.012987411580979824, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.7881759405136108, "rewards/total_composite/std": 0.04197310656309128, "reward": 0.7881759405136108, "reward_std": 0.04197310656309128, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18623779714107513, "sampling/sampling_logp_difference/max": 1.2195725440979004, "sampling/importance_sampling_ratio/min": 0.2953563928604126, "sampling/importance_sampling_ratio/mean": 1.0431245565414429, "sampling/importance_sampling_ratio/max": 1.80893075466156, "entropy": 2.054675593972206, "clip_ratio/low_mean": 0.045130813494324684, "clip_ratio/low_min": 0.045130813494324684, "clip_ratio/high_mean": 0.11178882885724306, "clip_ratio/high_max": 0.11178882885724306, "clip_ratio/region_mean": 0.15691964235156775, "reward_total_mean": 0.7881759405136108, "reward_meter_mean": 0.8690699934959412, "reward_meter_std": 0.1305169016122818, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9896947145462036, "reward_repeat_soft_std": 0.012987411580979824, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.7881759405136108, "reward_total_composite_std": 0.04197310656309128} {"timestamp_utc": "2026-04-12T23:21:52Z", "mode": "train", "global_step": 471, "epoch": 0.04731290808638875, "loss": 0.0295, "grad_norm": 4.50002384185791, "learning_rate": 8.575757575757575e-06, "num_tokens": 903189.0, "completions/mean_length": 133.0, "completions/min_length": 53.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 78.85714721679688, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 159.0, "rewards/meter/mean": 0.4656236171722412, "rewards/meter/std": 0.39359021186828613, "rewards/count_adherence/mean": 0.7916666865348816, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9904096126556396, "rewards/repeat_soft/std": 0.017097515985369682, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.20570002496242523, "rewards/total_composite/mean": 0.5015118718147278, "rewards/total_composite/std": 0.30027151107788086, "reward": 0.5015118718147278, "reward_std": 0.30027151107788086, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20438171923160553, "sampling/sampling_logp_difference/max": 0.854029655456543, "sampling/importance_sampling_ratio/min": 0.4256960451602936, "sampling/importance_sampling_ratio/mean": 1.055314540863037, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6479351818561554, "clip_ratio/low_mean": 0.04403852019459009, "clip_ratio/low_min": 0.04403852019459009, "clip_ratio/high_mean": 0.08991328440606594, "clip_ratio/high_max": 0.08991328440606594, "clip_ratio/region_mean": 0.13395180460065603, "reward_total_mean": 0.5015118718147278, "reward_meter_mean": 0.4656236171722412, "reward_meter_std": 0.39359021186828613, "reward_count_adherence_mean": 0.7916666865348816, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9904096126556396, "reward_repeat_soft_std": 0.017097515985369682, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.20570002496242523, "reward_total_composite_mean": 0.5015118718147278, "reward_total_composite_std": 0.30027151107788086} {"timestamp_utc": "2026-04-12T23:22:04Z", "mode": "train", "global_step": 472, "epoch": 0.04741336012054244, "loss": -0.1346, "grad_norm": 3.1242499351501465, "learning_rate": 8.572727272727274e-06, "num_tokens": 905082.0, "completions/mean_length": 189.625, "completions/min_length": 73.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 82.16667175292969, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.2221563458442688, "rewards/meter/std": 0.3285336196422577, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.24775780737400055, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9984360337257385, "rewards/repeat_soft/std": 0.001931598293595016, "rewards/judge_quality/mean": 0.2162500023841858, "rewards/judge_quality/std": 0.14879876375198364, "rewards/total_composite/mean": 0.3252120018005371, "rewards/total_composite/std": 0.2625366151332855, "reward": 0.3252120018005371, "reward_std": 0.26253658533096313, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2370249629020691, "sampling/sampling_logp_difference/max": 1.2314262390136719, "sampling/importance_sampling_ratio/min": 0.2918759882450104, "sampling/importance_sampling_ratio/mean": 1.0715183019638062, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.4178539514541626, "clip_ratio/low_mean": 0.024706458672881126, "clip_ratio/low_min": 0.024706458672881126, "clip_ratio/high_mean": 0.1010019239038229, "clip_ratio/high_max": 0.1010019239038229, "clip_ratio/region_mean": 0.12570838257670403, "reward_total_mean": 0.3252120018005371, "reward_meter_mean": 0.2221563458442688, "reward_meter_std": 0.3285336196422577, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.24775780737400055, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9984360337257385, "reward_repeat_soft_std": 0.001931598293595016, "reward_judge_quality_mean": 0.2162500023841858, "reward_judge_quality_std": 0.14879876375198364, "reward_total_composite_mean": 0.3252120018005371, "reward_total_composite_std": 0.2625366151332855} {"timestamp_utc": "2026-04-12T23:22:12Z", "mode": "train", "global_step": 473, "epoch": 0.04751381215469613, "loss": 0.0602, "grad_norm": 11.445880889892578, "learning_rate": 8.56969696969697e-06, "num_tokens": 906979.0, "completions/mean_length": 70.125, "completions/min_length": 50.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.125, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.29853105545043945, "rewards/meter/std": 0.3063596487045288, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9868161082267761, "rewards/repeat_soft/std": 0.013016137294471264, "rewards/judge_quality/mean": 0.36374998092651367, "rewards/judge_quality/std": 0.09500939399003983, "rewards/total_composite/mean": 0.4104326665401459, "rewards/total_composite/std": 0.2085748165845871, "reward": 0.4104326665401459, "reward_std": 0.2085747867822647, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.217630535364151, "sampling/sampling_logp_difference/max": 1.9158573150634766, "sampling/importance_sampling_ratio/min": 0.1472155749797821, "sampling/importance_sampling_ratio/mean": 1.0449424982070923, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5484287291765213, "clip_ratio/low_mean": 0.09479693695902824, "clip_ratio/low_min": 0.09479693695902824, "clip_ratio/high_mean": 0.09601440094411373, "clip_ratio/high_max": 0.09601440094411373, "clip_ratio/region_mean": 0.19081133790314198, "reward_total_mean": 0.4104326665401459, "reward_meter_mean": 0.29853105545043945, "reward_meter_std": 0.3063596487045288, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9868161082267761, "reward_repeat_soft_std": 0.013016137294471264, "reward_judge_quality_mean": 0.36374998092651367, "reward_judge_quality_std": 0.09500939399003983, "reward_total_composite_mean": 0.4104326665401459, "reward_total_composite_std": 0.2085748165845871} {"timestamp_utc": "2026-04-12T23:22:20Z", "mode": "train", "global_step": 474, "epoch": 0.04761426418884982, "loss": 0.0987, "grad_norm": 20.811996459960938, "learning_rate": 8.566666666666667e-06, "num_tokens": 908417.0, "completions/mean_length": 18.75, "completions/min_length": 16.0, "completions/max_length": 22.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 18.75, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 22.0, "rewards/meter/mean": 0.8286643028259277, "rewards/meter/std": 0.3347642719745636, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9566287994384766, "rewards/repeat_soft/std": 0.016606280580163002, "rewards/judge_quality/mean": 0.3725000023841858, "rewards/judge_quality/std": 0.11055056750774384, "rewards/total_composite/mean": 0.7303117513656616, "rewards/total_composite/std": 0.14762073755264282, "reward": 0.7303117513656616, "reward_std": 0.14762072265148163, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22031214833259583, "sampling/sampling_logp_difference/max": 1.1910629272460938, "sampling/importance_sampling_ratio/min": 0.30389806628227234, "sampling/importance_sampling_ratio/mean": 1.0562740564346313, "sampling/importance_sampling_ratio/max": 1.9779938459396362, "entropy": 2.7971817553043365, "clip_ratio/low_mean": 0.034988039173185825, "clip_ratio/low_min": 0.034988039173185825, "clip_ratio/high_mean": 0.1332933148369193, "clip_ratio/high_max": 0.1332933148369193, "clip_ratio/region_mean": 0.16828135401010513, "reward_total_mean": 0.7303117513656616, "reward_meter_mean": 0.8286643028259277, "reward_meter_std": 0.3347642719745636, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9566287994384766, "reward_repeat_soft_std": 0.016606280580163002, "reward_judge_quality_mean": 0.3725000023841858, "reward_judge_quality_std": 0.11055056750774384, "reward_total_composite_mean": 0.7303117513656616, "reward_total_composite_std": 0.14762073755264282} {"timestamp_utc": "2026-04-12T23:22:28Z", "mode": "train", "global_step": 475, "epoch": 0.047714716223003516, "loss": 0.0667, "grad_norm": 27.006237030029297, "learning_rate": 8.563636363636364e-06, "num_tokens": 909827.0, "completions/mean_length": 31.25, "completions/min_length": 28.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.25, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.31455662846565247, "rewards/meter/std": 0.23576343059539795, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9967487454414368, "rewards/repeat_soft/std": 0.006804719101637602, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.4783477187156677, "rewards/total_composite/std": 0.2395791858434677, "reward": 0.4783477187156677, "reward_std": 0.23957915604114532, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20534798502922058, "sampling/sampling_logp_difference/max": 2.551751136779785, "sampling/importance_sampling_ratio/min": 0.0779450535774231, "sampling/importance_sampling_ratio/mean": 1.0229233503341675, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9027165621519089, "clip_ratio/low_mean": 0.043926662765443325, "clip_ratio/low_min": 0.043926662765443325, "clip_ratio/high_mean": 0.11415474861860275, "clip_ratio/high_max": 0.11415474861860275, "clip_ratio/region_mean": 0.15808141138404608, "reward_total_mean": 0.4783477187156677, "reward_meter_mean": 0.31455662846565247, "reward_meter_std": 0.23576343059539795, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9967487454414368, "reward_repeat_soft_std": 0.006804719101637602, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.4783477187156677, "reward_total_composite_std": 0.2395791858434677} {"timestamp_utc": "2026-04-12T23:22:34Z", "mode": "train", "global_step": 476, "epoch": 0.04781516825715721, "loss": 0.0937, "grad_norm": 24.104907989501953, "learning_rate": 8.560606060606062e-06, "num_tokens": 911157.0, "completions/mean_length": 17.25, "completions/min_length": 15.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 17.25, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.8586819171905518, "rewards/meter/std": 0.2936249375343323, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9516729712486267, "rewards/repeat_soft/std": 0.02854108065366745, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404937028885, "rewards/total_composite/mean": 0.7984491586685181, "rewards/total_composite/std": 0.1614377647638321, "reward": 0.7984491586685181, "reward_std": 0.1614377498626709, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20040160417556763, "sampling/sampling_logp_difference/max": 1.6388707160949707, "sampling/importance_sampling_ratio/min": 0.19419921934604645, "sampling/importance_sampling_ratio/mean": 0.9797670841217041, "sampling/importance_sampling_ratio/max": 1.6464308500289917, "entropy": 1.8860146552324295, "clip_ratio/low_mean": 0.060421994887292385, "clip_ratio/low_min": 0.060421994887292385, "clip_ratio/high_mean": 0.11718750186264515, "clip_ratio/high_max": 0.11718750186264515, "clip_ratio/region_mean": 0.17760949674993753, "reward_total_mean": 0.7984491586685181, "reward_meter_mean": 0.8586819171905518, "reward_meter_std": 0.2936249375343323, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9516729712486267, "reward_repeat_soft_std": 0.02854108065366745, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404937028885, "reward_total_composite_mean": 0.7984491586685181, "reward_total_composite_std": 0.1614377647638321} {"timestamp_utc": "2026-04-12T23:22:41Z", "mode": "train", "global_step": 477, "epoch": 0.0479156202913109, "loss": -0.0275, "grad_norm": 14.953290939331055, "learning_rate": 8.557575757575757e-06, "num_tokens": 912755.0, "completions/mean_length": 33.75, "completions/min_length": 28.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.75, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.7389776706695557, "rewards/meter/std": 0.23934632539749146, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9966026544570923, "rewards/repeat_soft/std": 0.008344974368810654, "rewards/judge_quality/mean": 0.5575000047683716, "rewards/judge_quality/std": 0.19955310225486755, "rewards/total_composite/mean": 0.7494502067565918, "rewards/total_composite/std": 0.13745436072349548, "reward": 0.7494502067565918, "reward_std": 0.1374543309211731, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2166406661272049, "sampling/sampling_logp_difference/max": 1.4909486770629883, "sampling/importance_sampling_ratio/min": 0.2251589596271515, "sampling/importance_sampling_ratio/mean": 1.0131245851516724, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2558445781469345, "clip_ratio/low_mean": 0.0802703145891428, "clip_ratio/low_min": 0.0802703145891428, "clip_ratio/high_mean": 0.10306372866034508, "clip_ratio/high_max": 0.10306372866034508, "clip_ratio/region_mean": 0.18333404324948788, "reward_total_mean": 0.7494502067565918, "reward_meter_mean": 0.7389776706695557, "reward_meter_std": 0.23934632539749146, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9966026544570923, "reward_repeat_soft_std": 0.008344974368810654, "reward_judge_quality_mean": 0.5575000047683716, "reward_judge_quality_std": 0.19955310225486755, "reward_total_composite_mean": 0.7494502067565918, "reward_total_composite_std": 0.13745436072349548} {"timestamp_utc": "2026-04-12T23:22:53Z", "mode": "train", "global_step": 478, "epoch": 0.04801607232546459, "loss": -0.1228, "grad_norm": 2.2449262142181396, "learning_rate": 8.554545454545456e-06, "num_tokens": 914372.0, "completions/mean_length": 168.125, "completions/min_length": 44.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 53.5, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.49222588539123535, "rewards/meter/std": 0.3952081501483917, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9973024725914001, "rewards/repeat_soft/std": 0.00424335990101099, "rewards/judge_quality/mean": 0.3062499761581421, "rewards/judge_quality/std": 0.1686871498823166, "rewards/total_composite/mean": 0.49520936608314514, "rewards/total_composite/std": 0.3334261476993561, "reward": 0.49520936608314514, "reward_std": 0.3334261178970337, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23271967470645905, "sampling/sampling_logp_difference/max": 1.2145681381225586, "sampling/importance_sampling_ratio/min": 0.2968381941318512, "sampling/importance_sampling_ratio/mean": 1.0791785717010498, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.095309317111969, "clip_ratio/low_mean": 0.020454544574022293, "clip_ratio/low_min": 0.020454544574022293, "clip_ratio/high_mean": 0.1205546036362648, "clip_ratio/high_max": 0.1205546036362648, "clip_ratio/region_mean": 0.1410091482102871, "reward_total_mean": 0.49520936608314514, "reward_meter_mean": 0.49222588539123535, "reward_meter_std": 0.3952081501483917, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9973024725914001, "reward_repeat_soft_std": 0.00424335990101099, "reward_judge_quality_mean": 0.3062499761581421, "reward_judge_quality_std": 0.1686871498823166, "reward_total_composite_mean": 0.49520936608314514, "reward_total_composite_std": 0.3334261476993561} {"timestamp_utc": "2026-04-12T23:22:59Z", "mode": "train", "global_step": 479, "epoch": 0.04811652435961828, "loss": 0.0576, "grad_norm": 24.526464462280273, "learning_rate": 8.551515151515152e-06, "num_tokens": 915764.0, "completions/mean_length": 18.0, "completions/min_length": 16.0, "completions/max_length": 20.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 18.0, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 20.0, "rewards/meter/mean": 0.35410773754119873, "rewards/meter/std": 0.4204140305519104, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9552083015441895, "rewards/repeat_soft/std": 0.02062394842505455, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.5267443060874939, "rewards/total_composite/std": 0.19070421159267426, "reward": 0.5267443060874939, "reward_std": 0.19070421159267426, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21566160023212433, "sampling/sampling_logp_difference/max": 1.555060863494873, "sampling/importance_sampling_ratio/min": 0.2111765295267105, "sampling/importance_sampling_ratio/mean": 1.0216410160064697, "sampling/importance_sampling_ratio/max": 1.8743867874145508, "entropy": 1.9745263904333115, "clip_ratio/low_mean": 0.1606209184974432, "clip_ratio/low_min": 0.1606209184974432, "clip_ratio/high_mean": 0.08388157933950424, "clip_ratio/high_max": 0.08388157933950424, "clip_ratio/region_mean": 0.24450249783694744, "reward_total_mean": 0.5267443060874939, "reward_meter_mean": 0.35410773754119873, "reward_meter_std": 0.4204140305519104, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9552083015441895, "reward_repeat_soft_std": 0.02062394842505455, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.5267443060874939, "reward_total_composite_std": 0.19070421159267426} {"timestamp_utc": "2026-04-12T23:23:06Z", "mode": "train", "global_step": 480, "epoch": 0.048216976393771975, "loss": 0.0085, "grad_norm": 11.684322357177734, "learning_rate": 8.548484848484849e-06, "num_tokens": 917606.0, "completions/mean_length": 51.25, "completions/min_length": 45.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.25, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.6836015582084656, "rewards/meter/std": 0.3131708800792694, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.997057318687439, "rewards/repeat_soft/std": 0.00529851159080863, "rewards/judge_quality/mean": 0.4349999725818634, "rewards/judge_quality/std": 0.195521280169487, "rewards/total_composite/mean": 0.6878265142440796, "rewards/total_composite/std": 0.16177129745483398, "reward": 0.6878265142440796, "reward_std": 0.16177129745483398, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20774684846401215, "sampling/sampling_logp_difference/max": 1.188455581665039, "sampling/importance_sampling_ratio/min": 0.30469149351119995, "sampling/importance_sampling_ratio/mean": 1.045501708984375, "sampling/importance_sampling_ratio/max": 1.9677038192749023, "entropy": 2.639828532934189, "clip_ratio/low_mean": 0.046073718927800655, "clip_ratio/low_min": 0.046073718927800655, "clip_ratio/high_mean": 0.10942235589027405, "clip_ratio/high_max": 0.10942235589027405, "clip_ratio/region_mean": 0.1554960748180747, "reward_total_mean": 0.6878265142440796, "reward_meter_mean": 0.6836015582084656, "reward_meter_std": 0.3131708800792694, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.997057318687439, "reward_repeat_soft_std": 0.00529851159080863, "reward_judge_quality_mean": 0.4349999725818634, "reward_judge_quality_std": 0.195521280169487, "reward_total_composite_mean": 0.6878265142440796, "reward_total_composite_std": 0.16177129745483398} {"timestamp_utc": "2026-04-12T23:23:13Z", "mode": "train", "global_step": 481, "epoch": 0.04831742842792566, "loss": 0.0868, "grad_norm": 20.18193244934082, "learning_rate": 8.545454545454546e-06, "num_tokens": 918862.0, "completions/mean_length": 21.0, "completions/min_length": 16.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.0, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.8033577799797058, "rewards/meter/std": 0.3704932630062103, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.952572226524353, "rewards/repeat_soft/std": 0.0239147637039423, "rewards/judge_quality/mean": 0.5512499809265137, "rewards/judge_quality/std": 0.3296508193016052, "rewards/total_composite/mean": 0.7721432447433472, "rewards/total_composite/std": 0.22092854976654053, "reward": 0.7721432447433472, "reward_std": 0.22092856466770172, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18092982470989227, "sampling/sampling_logp_difference/max": 1.255601406097412, "sampling/importance_sampling_ratio/min": 0.28490445017814636, "sampling/importance_sampling_ratio/mean": 1.0527089834213257, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9545055478811264, "clip_ratio/low_mean": 0.057382246013730764, "clip_ratio/low_min": 0.057382246013730764, "clip_ratio/high_mean": 0.11703320685774088, "clip_ratio/high_max": 0.11703320685774088, "clip_ratio/region_mean": 0.17441545287147164, "reward_total_mean": 0.7721432447433472, "reward_meter_mean": 0.8033577799797058, "reward_meter_std": 0.3704932630062103, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.952572226524353, "reward_repeat_soft_std": 0.0239147637039423, "reward_judge_quality_mean": 0.5512499809265137, "reward_judge_quality_std": 0.3296508193016052, "reward_total_composite_mean": 0.7721432447433472, "reward_total_composite_std": 0.22092854976654053} {"timestamp_utc": "2026-04-12T23:23:20Z", "mode": "train", "global_step": 482, "epoch": 0.04841788046207936, "loss": 0.0741, "grad_norm": 13.236509323120117, "learning_rate": 8.542424242424243e-06, "num_tokens": 920651.0, "completions/mean_length": 44.625, "completions/min_length": 38.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.625, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.4346301555633545, "rewards/meter/std": 0.3646840751171112, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9881601333618164, "rewards/repeat_soft/std": 0.009196527302265167, "rewards/judge_quality/mean": 0.7400000095367432, "rewards/judge_quality/std": 0.14667749404907227, "rewards/total_composite/mean": 0.6663995385169983, "rewards/total_composite/std": 0.18745341897010803, "reward": 0.6663995385169983, "reward_std": 0.18745340406894684, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14564989507198334, "sampling/sampling_logp_difference/max": 2.1735196113586426, "sampling/importance_sampling_ratio/min": 0.11377646774053574, "sampling/importance_sampling_ratio/mean": 1.01572585105896, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0912136137485504, "clip_ratio/low_mean": 0.09795157331973314, "clip_ratio/low_min": 0.09795157331973314, "clip_ratio/high_mean": 0.07051282189786434, "clip_ratio/high_max": 0.07051282189786434, "clip_ratio/region_mean": 0.16846439521759748, "reward_total_mean": 0.6663995385169983, "reward_meter_mean": 0.4346301555633545, "reward_meter_std": 0.3646840751171112, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9881601333618164, "reward_repeat_soft_std": 0.009196527302265167, "reward_judge_quality_mean": 0.7400000095367432, "reward_judge_quality_std": 0.14667749404907227, "reward_total_composite_mean": 0.6663995385169983, "reward_total_composite_std": 0.18745341897010803} {"timestamp_utc": "2026-04-12T23:23:27Z", "mode": "train", "global_step": 483, "epoch": 0.04851833249623305, "loss": 0.1001, "grad_norm": 18.536949157714844, "learning_rate": 8.539393939393939e-06, "num_tokens": 922414.0, "completions/mean_length": 35.375, "completions/min_length": 31.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.375, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.7959853410720825, "rewards/meter/std": 0.3413451611995697, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9933829307556152, "rewards/repeat_soft/std": 0.013094097375869751, "rewards/judge_quality/mean": 0.4437499940395355, "rewards/judge_quality/std": 0.128834068775177, "rewards/total_composite/mean": 0.6568917036056519, "rewards/total_composite/std": 0.3077256679534912, "reward": 0.6568917036056519, "reward_std": 0.3077256381511688, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20516566932201385, "sampling/sampling_logp_difference/max": 1.2633590698242188, "sampling/importance_sampling_ratio/min": 0.28270280361175537, "sampling/importance_sampling_ratio/mean": 1.0518689155578613, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1065458804368973, "clip_ratio/low_mean": 0.03619186207652092, "clip_ratio/low_min": 0.03619186207652092, "clip_ratio/high_mean": 0.15277303289622068, "clip_ratio/high_max": 0.15277303289622068, "clip_ratio/region_mean": 0.1889648949727416, "reward_total_mean": 0.6568917036056519, "reward_meter_mean": 0.7959853410720825, "reward_meter_std": 0.3413451611995697, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9933829307556152, "reward_repeat_soft_std": 0.013094097375869751, "reward_judge_quality_mean": 0.4437499940395355, "reward_judge_quality_std": 0.128834068775177, "reward_total_composite_mean": 0.6568917036056519, "reward_total_composite_std": 0.3077256679534912} {"timestamp_utc": "2026-04-12T23:23:38Z", "mode": "train", "global_step": 484, "epoch": 0.04861878453038674, "loss": -0.0832, "grad_norm": 2.2935540676116943, "learning_rate": 8.536363636363636e-06, "num_tokens": 923964.0, "completions/mean_length": 156.75, "completions/min_length": 32.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 38.333335876464844, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.24783718585968018, "rewards/meter/std": 0.22763103246688843, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9979872703552246, "rewards/repeat_soft/std": 0.0033672908321022987, "rewards/judge_quality/mean": 0.3475000262260437, "rewards/judge_quality/std": 0.2776302695274353, "rewards/total_composite/mean": 0.35823217034339905, "rewards/total_composite/std": 0.24483224749565125, "reward": 0.35823217034339905, "reward_std": 0.24483221769332886, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.255837082862854, "sampling/sampling_logp_difference/max": 1.2945337295532227, "sampling/importance_sampling_ratio/min": 0.2740255892276764, "sampling/importance_sampling_ratio/mean": 1.0455281734466553, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.367824673652649, "clip_ratio/low_mean": 0.013888888992369175, "clip_ratio/low_min": 0.013888888992369175, "clip_ratio/high_mean": 0.14804302342236042, "clip_ratio/high_max": 0.14804302342236042, "clip_ratio/region_mean": 0.1619319124147296, "reward_total_mean": 0.35823217034339905, "reward_meter_mean": 0.24783718585968018, "reward_meter_std": 0.22763103246688843, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9979872703552246, "reward_repeat_soft_std": 0.0033672908321022987, "reward_judge_quality_mean": 0.3475000262260437, "reward_judge_quality_std": 0.2776302695274353, "reward_total_composite_mean": 0.35823217034339905, "reward_total_composite_std": 0.24483224749565125} {"timestamp_utc": "2026-04-12T23:23:44Z", "mode": "train", "global_step": 485, "epoch": 0.048719236564540434, "loss": 0.0556, "grad_norm": 38.08620071411133, "learning_rate": 8.533333333333335e-06, "num_tokens": 925344.0, "completions/mean_length": 22.5, "completions/min_length": 20.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.5, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.8514686822891235, "rewards/meter/std": 0.31843143701553345, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.7501608729362488, "rewards/total_composite/std": 0.14433911442756653, "reward": 0.7501608729362488, "reward_std": 0.14433911442756653, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23996224999427795, "sampling/sampling_logp_difference/max": 1.1786270141601562, "sampling/importance_sampling_ratio/min": 0.30770090222358704, "sampling/importance_sampling_ratio/mean": 1.0498254299163818, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.789001017808914, "clip_ratio/low_mean": 0.02595238061621785, "clip_ratio/low_min": 0.02595238061621785, "clip_ratio/high_mean": 0.09607323305681348, "clip_ratio/high_max": 0.09607323305681348, "clip_ratio/region_mean": 0.12202561367303133, "reward_total_mean": 0.7501608729362488, "reward_meter_mean": 0.8514686822891235, "reward_meter_std": 0.31843143701553345, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.7501608729362488, "reward_total_composite_std": 0.14433911442756653} {"timestamp_utc": "2026-04-12T23:23:55Z", "mode": "train", "global_step": 486, "epoch": 0.04881968859869412, "loss": -0.0428, "grad_norm": 4.27701473236084, "learning_rate": 8.53030303030303e-06, "num_tokens": 926708.0, "completions/mean_length": 82.5, "completions/min_length": 18.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 21.142858505249023, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.4715014100074768, "rewards/meter/std": 0.45728734135627747, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.3387500047683716, "rewards/judge_quality/std": 0.14327171444892883, "rewards/total_composite/mean": 0.47697019577026367, "rewards/total_composite/std": 0.3407522737979889, "reward": 0.47697019577026367, "reward_std": 0.3407522439956665, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21055954694747925, "sampling/sampling_logp_difference/max": 1.1676549911499023, "sampling/importance_sampling_ratio/min": 0.31109562516212463, "sampling/importance_sampling_ratio/mean": 1.0766068696975708, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3328903168439865, "clip_ratio/low_mean": 0.08884057961404324, "clip_ratio/low_min": 0.08884057961404324, "clip_ratio/high_mean": 0.07768640294671059, "clip_ratio/high_max": 0.07768640294671059, "clip_ratio/region_mean": 0.16652698256075382, "reward_total_mean": 0.47697019577026367, "reward_meter_mean": 0.4715014100074768, "reward_meter_std": 0.45728734135627747, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.3387500047683716, "reward_judge_quality_std": 0.14327171444892883, "reward_total_composite_mean": 0.47697019577026367, "reward_total_composite_std": 0.3407522737979889} {"timestamp_utc": "2026-04-12T23:24:01Z", "mode": "train", "global_step": 487, "epoch": 0.04892014063284782, "loss": 0.0707, "grad_norm": 18.004030227661133, "learning_rate": 8.527272727272728e-06, "num_tokens": 928441.0, "completions/mean_length": 36.625, "completions/min_length": 22.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.625, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.47886529564857483, "rewards/meter/std": 0.36675673723220825, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9964999556541443, "rewards/repeat_soft/std": 0.004299997817724943, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.5362805128097534, "rewards/total_composite/std": 0.2675591707229614, "reward": 0.5362805128097534, "reward_std": 0.2675591707229614, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1901964396238327, "sampling/sampling_logp_difference/max": 1.7129762172698975, "sampling/importance_sampling_ratio/min": 0.18032829463481903, "sampling/importance_sampling_ratio/mean": 0.9886579513549805, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.796732373535633, "clip_ratio/low_mean": 0.07079656631685793, "clip_ratio/low_min": 0.07079656631685793, "clip_ratio/high_mean": 0.07068077242001891, "clip_ratio/high_max": 0.07068077242001891, "clip_ratio/region_mean": 0.14147733873687685, "reward_total_mean": 0.5362805128097534, "reward_meter_mean": 0.47886529564857483, "reward_meter_std": 0.36675673723220825, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9964999556541443, "reward_repeat_soft_std": 0.004299997817724943, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.5362805128097534, "reward_total_composite_std": 0.2675591707229614} {"timestamp_utc": "2026-04-12T23:24:14Z", "mode": "train", "global_step": 488, "epoch": 0.049020592667001504, "loss": -0.1116, "grad_norm": 4.519741535186768, "learning_rate": 8.524242424242425e-06, "num_tokens": 930664.0, "completions/mean_length": 152.875, "completions/min_length": 95.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 101.5714340209961, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.8488990664482117, "rewards/meter/std": 0.23528046905994415, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9985690116882324, "rewards/repeat_soft/std": 0.001309814746491611, "rewards/judge_quality/mean": 0.2887499928474426, "rewards/judge_quality/std": 0.12799972295761108, "rewards/total_composite/mean": 0.5373495221138, "rewards/total_composite/std": 0.3521996736526489, "reward": 0.5373495221138, "reward_std": 0.3521996736526489, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2276471108198166, "sampling/sampling_logp_difference/max": 1.3377304077148438, "sampling/importance_sampling_ratio/min": 0.26244062185287476, "sampling/importance_sampling_ratio/mean": 1.0638020038604736, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.609214037656784, "clip_ratio/low_mean": 0.0420149527490139, "clip_ratio/low_min": 0.0420149527490139, "clip_ratio/high_mean": 0.111766685731709, "clip_ratio/high_max": 0.111766685731709, "clip_ratio/region_mean": 0.1537816384807229, "reward_total_mean": 0.5373495221138, "reward_meter_mean": 0.8488990664482117, "reward_meter_std": 0.23528046905994415, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9985690116882324, "reward_repeat_soft_std": 0.001309814746491611, "reward_judge_quality_mean": 0.2887499928474426, "reward_judge_quality_std": 0.12799972295761108, "reward_total_composite_mean": 0.5373495221138, "reward_total_composite_std": 0.3521996736526489} {"timestamp_utc": "2026-04-12T23:24:22Z", "mode": "train", "global_step": 489, "epoch": 0.0491210447011552, "loss": 0.0599, "grad_norm": 15.273518562316895, "learning_rate": 8.521212121212123e-06, "num_tokens": 932300.0, "completions/mean_length": 35.5, "completions/min_length": 23.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.5, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.731407642364502, "rewards/meter/std": 0.39545631408691406, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9558850526809692, "rewards/repeat_soft/std": 0.04194600135087967, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.720596969127655, "rewards/total_composite/std": 0.19984009861946106, "reward": 0.720596969127655, "reward_std": 0.19984008371829987, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17903836071491241, "sampling/sampling_logp_difference/max": 1.2575063705444336, "sampling/importance_sampling_ratio/min": 0.28436222672462463, "sampling/importance_sampling_ratio/mean": 1.0366971492767334, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5598444417119026, "clip_ratio/low_mean": 0.05457875691354275, "clip_ratio/low_min": 0.05457875691354275, "clip_ratio/high_mean": 0.16512398794293404, "clip_ratio/high_max": 0.16512398794293404, "clip_ratio/region_mean": 0.21970274485647678, "reward_total_mean": 0.720596969127655, "reward_meter_mean": 0.731407642364502, "reward_meter_std": 0.39545631408691406, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9558850526809692, "reward_repeat_soft_std": 0.04194600135087967, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.720596969127655, "reward_total_composite_std": 0.19984009861946106} {"timestamp_utc": "2026-04-12T23:24:29Z", "mode": "train", "global_step": 490, "epoch": 0.04922149673530889, "loss": 0.0258, "grad_norm": 22.8010311126709, "learning_rate": 8.518181818181818e-06, "num_tokens": 933659.0, "completions/mean_length": 21.875, "completions/min_length": 20.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.875, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.23559808731079102, "rewards/meter/std": 0.33045119047164917, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.954291582107544, "rewards/repeat_soft/std": 0.023216886445879936, "rewards/judge_quality/mean": 0.8025000095367432, "rewards/judge_quality/std": 0.21756774187088013, "rewards/total_composite/mean": 0.5921983122825623, "rewards/total_composite/std": 0.13734307885169983, "reward": 0.5921983122825623, "reward_std": 0.13734307885169983, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21041616797447205, "sampling/sampling_logp_difference/max": 1.4317502975463867, "sampling/importance_sampling_ratio/min": 0.23889043927192688, "sampling/importance_sampling_ratio/mean": 1.048363208770752, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2028914242982864, "clip_ratio/low_mean": 0.10670290095731616, "clip_ratio/low_min": 0.10670290095731616, "clip_ratio/high_mean": 0.07609265763312578, "clip_ratio/high_max": 0.07609265763312578, "clip_ratio/region_mean": 0.18279555859044194, "reward_total_mean": 0.5921983122825623, "reward_meter_mean": 0.23559808731079102, "reward_meter_std": 0.33045119047164917, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.954291582107544, "reward_repeat_soft_std": 0.023216886445879936, "reward_judge_quality_mean": 0.8025000095367432, "reward_judge_quality_std": 0.21756774187088013, "reward_total_composite_mean": 0.5921983122825623, "reward_total_composite_std": 0.13734307885169983} {"timestamp_utc": "2026-04-12T23:24:37Z", "mode": "train", "global_step": 491, "epoch": 0.04932194876946258, "loss": 0.0458, "grad_norm": 10.889242172241211, "learning_rate": 8.515151515151517e-06, "num_tokens": 935547.0, "completions/mean_length": 59.0, "completions/min_length": 52.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.0, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.6165602803230286, "rewards/meter/std": 0.2994098663330078, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9921756982803345, "rewards/repeat_soft/std": 0.008830866776406765, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.5534130930900574, "rewards/total_composite/std": 0.27373436093330383, "reward": 0.5534130930900574, "reward_std": 0.27373433113098145, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20469966530799866, "sampling/sampling_logp_difference/max": 1.4399423599243164, "sampling/importance_sampling_ratio/min": 0.2369414120912552, "sampling/importance_sampling_ratio/mean": 1.0349375009536743, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1396142542362213, "clip_ratio/low_mean": 0.07062262389808893, "clip_ratio/low_min": 0.07062262389808893, "clip_ratio/high_mean": 0.08466341905295849, "clip_ratio/high_max": 0.08466341905295849, "clip_ratio/region_mean": 0.15528604295104742, "reward_total_mean": 0.5534130930900574, "reward_meter_mean": 0.6165602803230286, "reward_meter_std": 0.2994098663330078, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9921756982803345, "reward_repeat_soft_std": 0.008830866776406765, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.5534130930900574, "reward_total_composite_std": 0.27373436093330383} {"timestamp_utc": "2026-04-12T23:24:48Z", "mode": "train", "global_step": 492, "epoch": 0.049422400803616276, "loss": 0.0021, "grad_norm": 5.519824981689453, "learning_rate": 8.512121212121213e-06, "num_tokens": 937445.0, "completions/mean_length": 125.25, "completions/min_length": 57.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 70.0, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.6111336946487427, "rewards/meter/std": 0.27438414096832275, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9858249425888062, "rewards/repeat_soft/std": 0.014319307170808315, "rewards/judge_quality/mean": 0.33124998211860657, "rewards/judge_quality/std": 0.13715866208076477, "rewards/total_composite/mean": 0.6167176961898804, "rewards/total_composite/std": 0.13357014954090118, "reward": 0.6167176961898804, "reward_std": 0.13357013463974, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21567761898040771, "sampling/sampling_logp_difference/max": 1.3624801635742188, "sampling/importance_sampling_ratio/min": 0.2560250163078308, "sampling/importance_sampling_ratio/mean": 1.063594102859497, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.722085326910019, "clip_ratio/low_mean": 0.07571574207395315, "clip_ratio/low_min": 0.07571574207395315, "clip_ratio/high_mean": 0.08312970027327538, "clip_ratio/high_max": 0.08312970027327538, "clip_ratio/region_mean": 0.15884544234722853, "reward_total_mean": 0.6167176961898804, "reward_meter_mean": 0.6111336946487427, "reward_meter_std": 0.27438414096832275, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9858249425888062, "reward_repeat_soft_std": 0.014319307170808315, "reward_judge_quality_mean": 0.33124998211860657, "reward_judge_quality_std": 0.13715866208076477, "reward_total_composite_mean": 0.6167176961898804, "reward_total_composite_std": 0.13357014954090118} {"timestamp_utc": "2026-04-12T23:24:54Z", "mode": "train", "global_step": 493, "epoch": 0.049522852837769964, "loss": -0.0257, "grad_norm": 16.08755111694336, "learning_rate": 8.50909090909091e-06, "num_tokens": 938785.0, "completions/mean_length": 23.5, "completions/min_length": 19.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.5, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.7254245281219482, "rewards/meter/std": 0.43751031160354614, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9532243013381958, "rewards/repeat_soft/std": 0.026235634461045265, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.22414520382881165, "rewards/total_composite/mean": 0.6664047241210938, "rewards/total_composite/std": 0.3125024437904358, "reward": 0.6664047241210938, "reward_std": 0.3125024437904358, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19488286972045898, "sampling/sampling_logp_difference/max": 1.6542072296142578, "sampling/importance_sampling_ratio/min": 0.19124361872673035, "sampling/importance_sampling_ratio/mean": 1.0465130805969238, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.98319311439991, "clip_ratio/low_mean": 0.05597826186567545, "clip_ratio/low_min": 0.05597826186567545, "clip_ratio/high_mean": 0.1306880209594965, "clip_ratio/high_max": 0.1306880209594965, "clip_ratio/region_mean": 0.18666628282517195, "reward_total_mean": 0.6664047241210938, "reward_meter_mean": 0.7254245281219482, "reward_meter_std": 0.43751031160354614, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9532243013381958, "reward_repeat_soft_std": 0.026235634461045265, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.22414520382881165, "reward_total_composite_mean": 0.6664047241210938, "reward_total_composite_std": 0.3125024437904358} {"timestamp_utc": "2026-04-12T23:25:00Z", "mode": "train", "global_step": 494, "epoch": 0.04962330487192366, "loss": 0.0779, "grad_norm": 12.544657707214355, "learning_rate": 8.506060606060607e-06, "num_tokens": 940658.0, "completions/mean_length": 69.125, "completions/min_length": 58.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.125, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.4183911681175232, "rewards/meter/std": 0.3719031512737274, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.992402970790863, "rewards/repeat_soft/std": 0.005193941295146942, "rewards/judge_quality/mean": 0.5362499952316284, "rewards/judge_quality/std": 0.1815754622220993, "rewards/total_composite/mean": 0.5983912944793701, "rewards/total_composite/std": 0.19607211649417877, "reward": 0.5983912944793701, "reward_std": 0.19607213139533997, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16405372321605682, "sampling/sampling_logp_difference/max": 1.302865982055664, "sampling/importance_sampling_ratio/min": 0.27175185084342957, "sampling/importance_sampling_ratio/mean": 1.0128458738327026, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2965889126062393, "clip_ratio/low_mean": 0.10305082891136408, "clip_ratio/low_min": 0.10305082891136408, "clip_ratio/high_mean": 0.06635014060884714, "clip_ratio/high_max": 0.06635014060884714, "clip_ratio/region_mean": 0.16940096952021122, "reward_total_mean": 0.5983912944793701, "reward_meter_mean": 0.4183911681175232, "reward_meter_std": 0.3719031512737274, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.992402970790863, "reward_repeat_soft_std": 0.005193941295146942, "reward_judge_quality_mean": 0.5362499952316284, "reward_judge_quality_std": 0.1815754622220993, "reward_total_composite_mean": 0.5983912944793701, "reward_total_composite_std": 0.19607211649417877} {"timestamp_utc": "2026-04-12T23:25:06Z", "mode": "train", "global_step": 495, "epoch": 0.049723756906077346, "loss": 0.0073, "grad_norm": 13.880448341369629, "learning_rate": 8.503030303030304e-06, "num_tokens": 942281.0, "completions/mean_length": 40.875, "completions/min_length": 38.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.875, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.591617226600647, "rewards/meter/std": 0.3035452663898468, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 1.0, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.5724999904632568, "rewards/judge_quality/std": 0.2129218727350235, "rewards/total_composite/mean": 0.6879777908325195, "rewards/total_composite/std": 0.09135298430919647, "reward": 0.6879777908325195, "reward_std": 0.09135296940803528, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1949581503868103, "sampling/sampling_logp_difference/max": 1.6147112846374512, "sampling/importance_sampling_ratio/min": 0.19894808530807495, "sampling/importance_sampling_ratio/mean": 1.0395232439041138, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9603798240423203, "clip_ratio/low_mean": 0.07617722544819117, "clip_ratio/low_min": 0.07617722544819117, "clip_ratio/high_mean": 0.08994528464972973, "clip_ratio/high_max": 0.08994528464972973, "clip_ratio/region_mean": 0.1661225100979209, "reward_total_mean": 0.6879777908325195, "reward_meter_mean": 0.591617226600647, "reward_meter_std": 0.3035452663898468, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 1.0, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.5724999904632568, "reward_judge_quality_std": 0.2129218727350235, "reward_total_composite_mean": 0.6879777908325195, "reward_total_composite_std": 0.09135298430919647} {"timestamp_utc": "2026-04-12T23:25:13Z", "mode": "train", "global_step": 496, "epoch": 0.04982420894023104, "loss": 0.0761, "grad_norm": 17.4720516204834, "learning_rate": 8.5e-06, "num_tokens": 943771.0, "completions/mean_length": 24.25, "completions/min_length": 19.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.25, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9627175331115723, "rewards/meter/std": 0.06472489982843399, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9439711570739746, "rewards/repeat_soft/std": 0.03057173267006874, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.805869996547699, "rewards/total_composite/std": 0.025754814967513084, "reward": 0.805869996547699, "reward_std": 0.025754820555448532, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16974593698978424, "sampling/sampling_logp_difference/max": 1.3749518394470215, "sampling/importance_sampling_ratio/min": 0.2528517544269562, "sampling/importance_sampling_ratio/mean": 1.0393706560134888, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.008498504757881, "clip_ratio/low_mean": 0.04418103490024805, "clip_ratio/low_min": 0.04418103490024805, "clip_ratio/high_mean": 0.13326327316462994, "clip_ratio/high_max": 0.13326327316462994, "clip_ratio/region_mean": 0.177444308064878, "reward_total_mean": 0.805869996547699, "reward_meter_mean": 0.9627175331115723, "reward_meter_std": 0.06472489982843399, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9439711570739746, "reward_repeat_soft_std": 0.03057173267006874, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.805869996547699, "reward_total_composite_std": 0.025754814967513084} {"timestamp_utc": "2026-04-12T23:25:24Z", "mode": "train", "global_step": 497, "epoch": 0.04992466097438473, "loss": -0.1489, "grad_norm": 2.4890153408050537, "learning_rate": 8.496969696969697e-06, "num_tokens": 945533.0, "completions/mean_length": 192.25, "completions/min_length": 75.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 85.66667175292969, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.5275389552116394, "rewards/meter/std": 0.3275412321090698, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9963271617889404, "rewards/repeat_soft/std": 0.00500798225402832, "rewards/judge_quality/mean": 0.29999998211860657, "rewards/judge_quality/std": 0.18007934093475342, "rewards/total_composite/mean": 0.45994633436203003, "rewards/total_composite/std": 0.32118988037109375, "reward": 0.45994633436203003, "reward_std": 0.32118988037109375, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21849308907985687, "sampling/sampling_logp_difference/max": 2.9372646808624268, "sampling/importance_sampling_ratio/min": 0.05301053449511528, "sampling/importance_sampling_ratio/mean": 1.0509265661239624, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9526351988315582, "clip_ratio/low_mean": 0.03981015086174011, "clip_ratio/low_min": 0.03981015086174011, "clip_ratio/high_mean": 0.09119764063507318, "clip_ratio/high_max": 0.09119764063507318, "clip_ratio/region_mean": 0.1310077914968133, "reward_total_mean": 0.45994633436203003, "reward_meter_mean": 0.5275389552116394, "reward_meter_std": 0.3275412321090698, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9963271617889404, "reward_repeat_soft_std": 0.00500798225402832, "reward_judge_quality_mean": 0.29999998211860657, "reward_judge_quality_std": 0.18007934093475342, "reward_total_composite_mean": 0.45994633436203003, "reward_total_composite_std": 0.32118988037109375} {"timestamp_utc": "2026-04-12T23:25:30Z", "mode": "train", "global_step": 498, "epoch": 0.05002511300853842, "loss": 0.0728, "grad_norm": 20.359010696411133, "learning_rate": 8.493939393939394e-06, "num_tokens": 946970.0, "completions/mean_length": 29.625, "completions/min_length": 21.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.625, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.5493113994598389, "rewards/meter/std": 0.4333387315273285, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9883812665939331, "rewards/repeat_soft/std": 0.014103373512625694, "rewards/judge_quality/mean": 0.5099999904632568, "rewards/judge_quality/std": 0.26586785912513733, "rewards/total_composite/mean": 0.6490283012390137, "rewards/total_composite/std": 0.23951449990272522, "reward": 0.6490283012390137, "reward_std": 0.2395145148038864, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1788337081670761, "sampling/sampling_logp_difference/max": 1.7278800010681152, "sampling/importance_sampling_ratio/min": 0.17766065895557404, "sampling/importance_sampling_ratio/mean": 1.0155634880065918, "sampling/importance_sampling_ratio/max": 1.8837411403656006, "entropy": 1.7680440098047256, "clip_ratio/low_mean": 0.05866630282253027, "clip_ratio/low_min": 0.05866630282253027, "clip_ratio/high_mean": 0.08523403853178024, "clip_ratio/high_max": 0.08523403853178024, "clip_ratio/region_mean": 0.1439003413543105, "reward_total_mean": 0.6490283012390137, "reward_meter_mean": 0.5493113994598389, "reward_meter_std": 0.4333387315273285, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9883812665939331, "reward_repeat_soft_std": 0.014103373512625694, "reward_judge_quality_mean": 0.5099999904632568, "reward_judge_quality_std": 0.26586785912513733, "reward_total_composite_mean": 0.6490283012390137, "reward_total_composite_std": 0.23951449990272522} {"timestamp_utc": "2026-04-12T23:25:36Z", "mode": "train", "global_step": 499, "epoch": 0.05012556504269212, "loss": 0.0894, "grad_norm": 13.443181991577148, "learning_rate": 8.490909090909092e-06, "num_tokens": 948534.0, "completions/mean_length": 48.5, "completions/min_length": 41.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.5, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.673080325126648, "rewards/meter/std": 0.3881789445877075, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9960678815841675, "rewards/repeat_soft/std": 0.007661442272365093, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.5987876653671265, "rewards/total_composite/std": 0.3006487190723419, "reward": 0.5987876653671265, "reward_std": 0.30064868927001953, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18640702962875366, "sampling/sampling_logp_difference/max": 1.99531888961792, "sampling/importance_sampling_ratio/min": 0.13597029447555542, "sampling/importance_sampling_ratio/mean": 0.9948229789733887, "sampling/importance_sampling_ratio/max": 1.7597625255584717, "entropy": 2.0359032601118088, "clip_ratio/low_mean": 0.056156616657972336, "clip_ratio/low_min": 0.056156616657972336, "clip_ratio/high_mean": 0.09623762872070074, "clip_ratio/high_max": 0.09623762872070074, "clip_ratio/region_mean": 0.15239424537867308, "reward_total_mean": 0.5987876653671265, "reward_meter_mean": 0.673080325126648, "reward_meter_std": 0.3881789445877075, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9960678815841675, "reward_repeat_soft_std": 0.007661442272365093, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.5987876653671265, "reward_total_composite_std": 0.3006487190723419} {"timestamp_utc": "2026-04-12T23:25:43Z", "mode": "train", "global_step": 500, "epoch": 0.050226017076845805, "loss": -0.0036, "grad_norm": 9.132417678833008, "learning_rate": 8.487878787878789e-06, "num_tokens": 950652.0, "completions/mean_length": 90.75, "completions/min_length": 73.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.75, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.4744132161140442, "rewards/meter/std": 0.29868823289871216, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9976946711540222, "rewards/repeat_soft/std": 0.003002408193424344, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.14520922303199768, "rewards/total_composite/mean": 0.5877553820610046, "rewards/total_composite/std": 0.13328135013580322, "reward": 0.5877553820610046, "reward_std": 0.13328132033348083, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18936270475387573, "sampling/sampling_logp_difference/max": 1.2891731262207031, "sampling/importance_sampling_ratio/min": 0.27549847960472107, "sampling/importance_sampling_ratio/mean": 1.0649367570877075, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5455402731895447, "clip_ratio/low_mean": 0.09222470410168171, "clip_ratio/low_min": 0.09222470410168171, "clip_ratio/high_mean": 0.07126624044030905, "clip_ratio/high_max": 0.07126624044030905, "clip_ratio/region_mean": 0.16349094454199076, "reward_total_mean": 0.5877553820610046, "reward_meter_mean": 0.4744132161140442, "reward_meter_std": 0.29868823289871216, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9976946711540222, "reward_repeat_soft_std": 0.003002408193424344, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.14520922303199768, "reward_total_composite_mean": 0.5877553820610046, "reward_total_composite_std": 0.13328135013580322} {"timestamp_utc": "2026-04-12T23:26:36Z", "mode": "eval", "global_step": 500, "epoch": 0.050226017076845805, "eval_loss": NaN, "eval_runtime": 52.9289, "eval_samples_per_second": 1.511, "eval_steps_per_second": 0.189, "eval_num_tokens": 950652.0, "eval_completions/mean_length": 79.7125, "eval_completions/min_length": 28.8, "eval_completions/max_length": 231.2, "eval_completions/clipped_ratio": 0.0375, "eval_completions/mean_terminated_length": 62.97321548461914, "eval_completions/min_terminated_length": 28.8, "eval_completions/max_terminated_length": 111.0, "eval_rewards/meter/mean": 0.6309856623411179, "eval_rewards/meter/std": 0.3115048572421074, "eval_rewards/count_adherence/mean": 0.9833333194255829, "eval_rewards/count_adherence/std": 0.04307050965726376, "eval_rewards/hard_gate/mean": 0.9375, "eval_rewards/hard_gate/std": 0.15235702097415924, "eval_rewards/repeat_soft/mean": 0.9879626631736755, "eval_rewards/repeat_soft/std": 0.011814793571829796, "eval_rewards/judge_quality/mean": 0.4897500067949295, "eval_rewards/judge_quality/std": 0.20685324743390082, "eval_rewards/total_composite/mean": 0.6499115228652954, "eval_rewards/total_composite/std": 0.21378921791911126, "eval_reward": 0.6499115228652954, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.1382225386798382, "eval_sampling/sampling_logp_difference/max": 1.0124813556671142, "eval_sampling/importance_sampling_ratio/min": 0.3746115654706955, "eval_sampling/importance_sampling_ratio/mean": 1.0481449961662292, "eval_sampling/importance_sampling_ratio/max": 1.5556179523468017, "eval_entropy": 2.038919460773468, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6499115228652954, "eval_reward_meter_mean": 0.6309856623411179, "eval_reward_meter_std": 0.3115048572421074, "eval_reward_count_adherence_mean": 0.9833333194255829, "eval_reward_count_adherence_std": 0.04307050965726376, "eval_reward_hard_gate_mean": 0.9375, "eval_reward_hard_gate_std": 0.15235702097415924, "eval_reward_repeat_soft_mean": 0.9879626631736755, "eval_reward_repeat_soft_std": 0.011814793571829796, "eval_reward_judge_quality_mean": 0.4897500067949295, "eval_reward_judge_quality_std": 0.20685324743390082, "eval_reward_total_composite_mean": 0.6499115228652954, "eval_reward_total_composite_std": 0.21378921791911126} {"timestamp_utc": "2026-04-12T23:26:50Z", "mode": "train", "global_step": 501, "epoch": 0.0503264691109995, "loss": -0.0684, "grad_norm": 4.764578819274902, "learning_rate": 8.484848484848486e-06, "num_tokens": 952364.0, "completions/mean_length": 109.0, "completions/min_length": 45.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 51.42857360839844, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.4882221817970276, "rewards/meter/std": 0.3284251093864441, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.993233323097229, "rewards/repeat_soft/std": 0.004344129003584385, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.18845234811306, "rewards/total_composite/mean": 0.48262131214141846, "rewards/total_composite/std": 0.3239957392215729, "reward": 0.48262131214141846, "reward_std": 0.3239957094192505, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20700344443321228, "sampling/sampling_logp_difference/max": 2.45961856842041, "sampling/importance_sampling_ratio/min": 0.08546754717826843, "sampling/importance_sampling_ratio/mean": 1.0494290590286255, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.033759757876396, "clip_ratio/low_mean": 0.05227272771298885, "clip_ratio/low_min": 0.05227272771298885, "clip_ratio/high_mean": 0.1278796847909689, "clip_ratio/high_max": 0.1278796847909689, "clip_ratio/region_mean": 0.18015241250395775, "reward_total_mean": 0.48262131214141846, "reward_meter_mean": 0.4882221817970276, "reward_meter_std": 0.3284251093864441, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.993233323097229, "reward_repeat_soft_std": 0.004344129003584385, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.18845234811306, "reward_total_composite_mean": 0.48262131214141846, "reward_total_composite_std": 0.3239957392215729} {"timestamp_utc": "2026-04-12T23:26:57Z", "mode": "train", "global_step": 502, "epoch": 0.05042692114515319, "loss": 0.0741, "grad_norm": 10.360894203186035, "learning_rate": 8.481818181818182e-06, "num_tokens": 954607.0, "completions/mean_length": 92.375, "completions/min_length": 67.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.375, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.44848525524139404, "rewards/meter/std": 0.26260483264923096, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9983288645744324, "rewards/repeat_soft/std": 0.001114478218369186, "rewards/judge_quality/mean": 0.5862499475479126, "rewards/judge_quality/std": 0.2500821650028229, "rewards/total_composite/mean": 0.6228387355804443, "rewards/total_composite/std": 0.14896026253700256, "reward": 0.6228387355804443, "reward_std": 0.14896026253700256, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17659489810466766, "sampling/sampling_logp_difference/max": 1.6058359146118164, "sampling/importance_sampling_ratio/min": 0.20072169601917267, "sampling/importance_sampling_ratio/mean": 1.0256593227386475, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9759622812271118, "clip_ratio/low_mean": 0.09438210632652044, "clip_ratio/low_min": 0.09438210632652044, "clip_ratio/high_mean": 0.05671723745763302, "clip_ratio/high_max": 0.05671723745763302, "clip_ratio/region_mean": 0.15109934378415346, "reward_total_mean": 0.6228387355804443, "reward_meter_mean": 0.44848525524139404, "reward_meter_std": 0.26260483264923096, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9983288645744324, "reward_repeat_soft_std": 0.001114478218369186, "reward_judge_quality_mean": 0.5862499475479126, "reward_judge_quality_std": 0.2500821650028229, "reward_total_composite_mean": 0.6228387355804443, "reward_total_composite_std": 0.14896026253700256} {"timestamp_utc": "2026-04-12T23:27:08Z", "mode": "train", "global_step": 503, "epoch": 0.05052737317930688, "loss": -0.1834, "grad_norm": 2.373995780944824, "learning_rate": 8.478787878787879e-06, "num_tokens": 956744.0, "completions/mean_length": 203.125, "completions/min_length": 93.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 100.16667175292969, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.7924829125404358, "rewards/meter/std": 0.2902827560901642, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.18600596487522125, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9907145500183105, "rewards/repeat_soft/std": 0.013672076165676117, "rewards/judge_quality/mean": 0.36000001430511475, "rewards/judge_quality/std": 0.2626241147518158, "rewards/total_composite/mean": 0.5725976228713989, "rewards/total_composite/std": 0.38128662109375, "reward": 0.5725976228713989, "reward_std": 0.3812865912914276, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2038375437259674, "sampling/sampling_logp_difference/max": 1.2044105529785156, "sampling/importance_sampling_ratio/min": 0.29986870288848877, "sampling/importance_sampling_ratio/mean": 1.060585618019104, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.118591845035553, "clip_ratio/low_mean": 0.015306122601032257, "clip_ratio/low_min": 0.015306122601032257, "clip_ratio/high_mean": 0.1074887067079544, "clip_ratio/high_max": 0.1074887067079544, "clip_ratio/region_mean": 0.12279482930898666, "reward_total_mean": 0.5725976228713989, "reward_meter_mean": 0.7924829125404358, "reward_meter_std": 0.2902827560901642, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.18600596487522125, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9907145500183105, "reward_repeat_soft_std": 0.013672076165676117, "reward_judge_quality_mean": 0.36000001430511475, "reward_judge_quality_std": 0.2626241147518158, "reward_total_composite_mean": 0.5725976228713989, "reward_total_composite_std": 0.38128662109375} {"timestamp_utc": "2026-04-12T23:27:14Z", "mode": "train", "global_step": 504, "epoch": 0.05062782521346057, "loss": 0.069, "grad_norm": 21.59787368774414, "learning_rate": 8.475757575757576e-06, "num_tokens": 958244.0, "completions/mean_length": 26.5, "completions/min_length": 19.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.5, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.9790815114974976, "rewards/meter/std": 0.0313568077981472, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9475217461585999, "rewards/repeat_soft/std": 0.042364854365587234, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.8072138428688049, "rewards/total_composite/std": 0.027990199625492096, "reward": 0.8072138428688049, "reward_std": 0.0279901921749115, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16848190128803253, "sampling/sampling_logp_difference/max": 1.0923690795898438, "sampling/importance_sampling_ratio/min": 0.3354209065437317, "sampling/importance_sampling_ratio/mean": 1.0355780124664307, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5157913193106651, "clip_ratio/low_mean": 0.038095238618552685, "clip_ratio/low_min": 0.038095238618552685, "clip_ratio/high_mean": 0.10272034350782633, "clip_ratio/high_max": 0.10272034350782633, "clip_ratio/region_mean": 0.140815582126379, "reward_total_mean": 0.8072138428688049, "reward_meter_mean": 0.9790815114974976, "reward_meter_std": 0.0313568077981472, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9475217461585999, "reward_repeat_soft_std": 0.042364854365587234, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.8072138428688049, "reward_total_composite_std": 0.027990199625492096} {"timestamp_utc": "2026-04-12T23:27:20Z", "mode": "train", "global_step": 505, "epoch": 0.050728277247614265, "loss": 0.0082, "grad_norm": 14.043022155761719, "learning_rate": 8.472727272727274e-06, "num_tokens": 959807.0, "completions/mean_length": 40.375, "completions/min_length": 35.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.375, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.929308295249939, "rewards/meter/std": 0.15854625403881073, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9899916648864746, "rewards/repeat_soft/std": 0.013865159824490547, "rewards/judge_quality/mean": 0.45499998331069946, "rewards/judge_quality/std": 0.12739142775535583, "rewards/total_composite/mean": 0.8036879301071167, "rewards/total_composite/std": 0.08234308660030365, "reward": 0.8036879301071167, "reward_std": 0.08234308660030365, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19629181921482086, "sampling/sampling_logp_difference/max": 1.1834602355957031, "sampling/importance_sampling_ratio/min": 0.3062173128128052, "sampling/importance_sampling_ratio/mean": 1.0416852235794067, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.295401930809021, "clip_ratio/low_mean": 0.05204116180539131, "clip_ratio/low_min": 0.05204116180539131, "clip_ratio/high_mean": 0.12614360358566046, "clip_ratio/high_max": 0.12614360358566046, "clip_ratio/region_mean": 0.17818476539105177, "reward_total_mean": 0.8036879301071167, "reward_meter_mean": 0.929308295249939, "reward_meter_std": 0.15854625403881073, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9899916648864746, "reward_repeat_soft_std": 0.013865159824490547, "reward_judge_quality_mean": 0.45499998331069946, "reward_judge_quality_std": 0.12739142775535583, "reward_total_composite_mean": 0.8036879301071167, "reward_total_composite_std": 0.08234308660030365} {"timestamp_utc": "2026-04-12T23:27:26Z", "mode": "train", "global_step": 506, "epoch": 0.05082872928176796, "loss": 0.0656, "grad_norm": 15.656820297241211, "learning_rate": 8.46969696969697e-06, "num_tokens": 961263.0, "completions/mean_length": 32.0, "completions/min_length": 29.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.8528526425361633, "rewards/meter/std": 0.29942867159843445, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9938091039657593, "rewards/repeat_soft/std": 0.007811932824552059, "rewards/judge_quality/mean": 0.71875, "rewards/judge_quality/std": 0.23805388808250427, "rewards/total_composite/mean": 0.8487895727157593, "rewards/total_composite/std": 0.18078812956809998, "reward": 0.8487895727157593, "reward_std": 0.18078814446926117, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11043333262205124, "sampling/sampling_logp_difference/max": 0.9747066497802734, "sampling/importance_sampling_ratio/min": 0.37730303406715393, "sampling/importance_sampling_ratio/mean": 1.0223491191864014, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6342151165008545, "clip_ratio/low_mean": 0.08498237933963537, "clip_ratio/low_min": 0.08498237933963537, "clip_ratio/high_mean": 0.027763118501752615, "clip_ratio/high_max": 0.027763118501752615, "clip_ratio/region_mean": 0.11274549784138799, "reward_total_mean": 0.8487895727157593, "reward_meter_mean": 0.8528526425361633, "reward_meter_std": 0.29942867159843445, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9938091039657593, "reward_repeat_soft_std": 0.007811932824552059, "reward_judge_quality_mean": 0.71875, "reward_judge_quality_std": 0.23805388808250427, "reward_total_composite_mean": 0.8487895727157593, "reward_total_composite_std": 0.18078812956809998} {"timestamp_utc": "2026-04-12T23:27:32Z", "mode": "train", "global_step": 507, "epoch": 0.05092918131592165, "loss": 0.0485, "grad_norm": 14.929595947265625, "learning_rate": 8.466666666666668e-06, "num_tokens": 962670.0, "completions/mean_length": 27.875, "completions/min_length": 20.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.875, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.8158579468727112, "rewards/meter/std": 0.22495533525943756, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9899121522903442, "rewards/repeat_soft/std": 0.02065182477235794, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.8358772993087769, "rewards/total_composite/std": 0.11943580955266953, "reward": 0.8358772993087769, "reward_std": 0.11943580210208893, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14432023465633392, "sampling/sampling_logp_difference/max": 1.0051093101501465, "sampling/importance_sampling_ratio/min": 0.3660046458244324, "sampling/importance_sampling_ratio/mean": 1.0389881134033203, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.235595241189003, "clip_ratio/low_mean": 0.09632024355232716, "clip_ratio/low_min": 0.09632024355232716, "clip_ratio/high_mean": 0.06456653214991093, "clip_ratio/high_max": 0.06456653214991093, "clip_ratio/region_mean": 0.16088677570223808, "reward_total_mean": 0.8358772993087769, "reward_meter_mean": 0.8158579468727112, "reward_meter_std": 0.22495533525943756, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9899121522903442, "reward_repeat_soft_std": 0.02065182477235794, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.8358772993087769, "reward_total_composite_std": 0.11943580955266953} {"timestamp_utc": "2026-04-12T23:27:39Z", "mode": "train", "global_step": 508, "epoch": 0.05102963335007534, "loss": 0.0163, "grad_norm": 13.003607749938965, "learning_rate": 8.463636363636364e-06, "num_tokens": 964497.0, "completions/mean_length": 63.375, "completions/min_length": 50.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.375, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.8074355721473694, "rewards/meter/std": 0.16525107622146606, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.995194673538208, "rewards/repeat_soft/std": 0.004917542915791273, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.6593576073646545, "rewards/total_composite/std": 0.27471402287483215, "reward": 0.6593576073646545, "reward_std": 0.27471399307250977, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18589268624782562, "sampling/sampling_logp_difference/max": 1.4275217056274414, "sampling/importance_sampling_ratio/min": 0.23990273475646973, "sampling/importance_sampling_ratio/mean": 1.0347625017166138, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8959117233753204, "clip_ratio/low_mean": 0.04172932356595993, "clip_ratio/low_min": 0.04172932356595993, "clip_ratio/high_mean": 0.123478296212852, "clip_ratio/high_max": 0.123478296212852, "clip_ratio/region_mean": 0.16520761977881193, "reward_total_mean": 0.6593576073646545, "reward_meter_mean": 0.8074355721473694, "reward_meter_std": 0.16525107622146606, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.995194673538208, "reward_repeat_soft_std": 0.004917542915791273, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.6593576073646545, "reward_total_composite_std": 0.27471402287483215} {"timestamp_utc": "2026-04-12T23:27:45Z", "mode": "train", "global_step": 509, "epoch": 0.05113008538422903, "loss": 0.0119, "grad_norm": 14.38261604309082, "learning_rate": 8.460606060606061e-06, "num_tokens": 966062.0, "completions/mean_length": 43.625, "completions/min_length": 38.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.625, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.9174467921257019, "rewards/meter/std": 0.1536252349615097, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.999160885810852, "rewards/repeat_soft/std": 0.002373373368754983, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.12906257808208466, "rewards/total_composite/mean": 0.7947671413421631, "rewards/total_composite/std": 0.08212077617645264, "reward": 0.7947671413421631, "reward_std": 0.08212078362703323, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21435587108135223, "sampling/sampling_logp_difference/max": 1.2104566097259521, "sampling/importance_sampling_ratio/min": 0.29806116223335266, "sampling/importance_sampling_ratio/mean": 1.0098544359207153, "sampling/importance_sampling_ratio/max": 1.9573719501495361, "entropy": 2.471867322921753, "clip_ratio/low_mean": 0.05346401780843735, "clip_ratio/low_min": 0.05346401780843735, "clip_ratio/high_mean": 0.1539918538182974, "clip_ratio/high_max": 0.1539918538182974, "clip_ratio/region_mean": 0.20745587162673473, "reward_total_mean": 0.7947671413421631, "reward_meter_mean": 0.9174467921257019, "reward_meter_std": 0.1536252349615097, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.999160885810852, "reward_repeat_soft_std": 0.002373373368754983, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.12906257808208466, "reward_total_composite_mean": 0.7947671413421631, "reward_total_composite_std": 0.08212077617645264} {"timestamp_utc": "2026-04-12T23:27:52Z", "mode": "train", "global_step": 510, "epoch": 0.051230537418382724, "loss": 0.2577, "grad_norm": 11.705020904541016, "learning_rate": 8.457575757575758e-06, "num_tokens": 967992.0, "completions/mean_length": 73.25, "completions/min_length": 51.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.25, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.6068181395530701, "rewards/meter/std": 0.3324304223060608, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9932630062103271, "rewards/repeat_soft/std": 0.007693739607930183, "rewards/judge_quality/mean": 0.4737499952316284, "rewards/judge_quality/std": 0.16291433572769165, "rewards/total_composite/mean": 0.645769476890564, "rewards/total_composite/std": 0.18117788434028625, "reward": 0.645769476890564, "reward_std": 0.18117788434028625, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2009975165128708, "sampling/sampling_logp_difference/max": 1.437356948852539, "sampling/importance_sampling_ratio/min": 0.23755480349063873, "sampling/importance_sampling_ratio/mean": 1.0444713830947876, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.542813301086426, "clip_ratio/low_mean": 0.04787276918068528, "clip_ratio/low_min": 0.04787276918068528, "clip_ratio/high_mean": 0.12310670129954815, "clip_ratio/high_max": 0.12310670129954815, "clip_ratio/region_mean": 0.17097947048023343, "reward_total_mean": 0.645769476890564, "reward_meter_mean": 0.6068181395530701, "reward_meter_std": 0.3324304223060608, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9932630062103271, "reward_repeat_soft_std": 0.007693739607930183, "reward_judge_quality_mean": 0.4737499952316284, "reward_judge_quality_std": 0.16291433572769165, "reward_total_composite_mean": 0.645769476890564, "reward_total_composite_std": 0.18117788434028625} {"timestamp_utc": "2026-04-12T23:27:58Z", "mode": "train", "global_step": 511, "epoch": 0.05133098945253641, "loss": 0.1071, "grad_norm": 19.793907165527344, "learning_rate": 8.454545454545455e-06, "num_tokens": 969598.0, "completions/mean_length": 42.75, "completions/min_length": 35.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.75, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.8164535164833069, "rewards/meter/std": 0.3201797306537628, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9806473255157471, "rewards/repeat_soft/std": 0.037479218095541, "rewards/judge_quality/mean": 0.5112500190734863, "rewards/judge_quality/std": 0.268936425447464, "rewards/total_composite/mean": 0.7688437700271606, "rewards/total_composite/std": 0.17216511070728302, "reward": 0.7688437700271606, "reward_std": 0.17216509580612183, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17595110833644867, "sampling/sampling_logp_difference/max": 1.3333244323730469, "sampling/importance_sampling_ratio/min": 0.2635994851589203, "sampling/importance_sampling_ratio/mean": 1.0398532152175903, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7407528012990952, "clip_ratio/low_mean": 0.06341874226927757, "clip_ratio/low_min": 0.06341874226927757, "clip_ratio/high_mean": 0.10338093992322683, "clip_ratio/high_max": 0.10338093992322683, "clip_ratio/region_mean": 0.1667996821925044, "reward_total_mean": 0.7688437700271606, "reward_meter_mean": 0.8164535164833069, "reward_meter_std": 0.3201797306537628, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9806473255157471, "reward_repeat_soft_std": 0.037479218095541, "reward_judge_quality_mean": 0.5112500190734863, "reward_judge_quality_std": 0.268936425447464, "reward_total_composite_mean": 0.7688437700271606, "reward_total_composite_std": 0.17216511070728302} {"timestamp_utc": "2026-04-12T23:28:04Z", "mode": "train", "global_step": 512, "epoch": 0.051431441486690106, "loss": 0.0373, "grad_norm": 14.056915283203125, "learning_rate": 8.451515151515151e-06, "num_tokens": 971109.0, "completions/mean_length": 31.875, "completions/min_length": 28.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.875, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.7269572019577026, "rewards/meter/std": 0.22113606333732605, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9959856271743774, "rewards/repeat_soft/std": 0.00792237464338541, "rewards/judge_quality/mean": 0.7312500476837158, "rewards/judge_quality/std": 0.2623213827610016, "rewards/total_composite/mean": 0.7253631949424744, "rewards/total_composite/std": 0.3065732717514038, "reward": 0.7253631949424744, "reward_std": 0.3065732717514038, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11145146191120148, "sampling/sampling_logp_difference/max": 2.4717180728912354, "sampling/importance_sampling_ratio/min": 0.08443965762853622, "sampling/importance_sampling_ratio/mean": 0.991574764251709, "sampling/importance_sampling_ratio/max": 1.967185378074646, "entropy": 0.3961819652467966, "clip_ratio/low_mean": 0.015269886702299118, "clip_ratio/low_min": 0.015269886702299118, "clip_ratio/high_mean": 0.08594991592690349, "clip_ratio/high_max": 0.08594991592690349, "clip_ratio/region_mean": 0.1012198026292026, "reward_total_mean": 0.7253631949424744, "reward_meter_mean": 0.7269572019577026, "reward_meter_std": 0.22113606333732605, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9959856271743774, "reward_repeat_soft_std": 0.00792237464338541, "reward_judge_quality_mean": 0.7312500476837158, "reward_judge_quality_std": 0.2623213827610016, "reward_total_composite_mean": 0.7253631949424744, "reward_total_composite_std": 0.3065732717514038} {"timestamp_utc": "2026-04-12T23:28:11Z", "mode": "train", "global_step": 513, "epoch": 0.051531893520843794, "loss": 0.0706, "grad_norm": 10.117542266845703, "learning_rate": 8.44848484848485e-06, "num_tokens": 973510.0, "completions/mean_length": 107.125, "completions/min_length": 86.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.125, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.8818994164466858, "rewards/meter/std": 0.22519220411777496, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9731980562210083, "rewards/repeat_soft/std": 0.026073496788740158, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.7158988118171692, "rewards/total_composite/std": 0.2906969487667084, "reward": 0.7158988118171692, "reward_std": 0.2906968891620636, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20275002717971802, "sampling/sampling_logp_difference/max": 1.4544191360473633, "sampling/importance_sampling_ratio/min": 0.2335359752178192, "sampling/importance_sampling_ratio/mean": 1.057955026626587, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6251766681671143, "clip_ratio/low_mean": 0.013319672085344791, "clip_ratio/low_min": 0.013319672085344791, "clip_ratio/high_mean": 0.16146229207515717, "clip_ratio/high_max": 0.16146229207515717, "clip_ratio/region_mean": 0.17478196416050196, "reward_total_mean": 0.7158988118171692, "reward_meter_mean": 0.8818994164466858, "reward_meter_std": 0.22519220411777496, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9731980562210083, "reward_repeat_soft_std": 0.026073496788740158, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.7158988118171692, "reward_total_composite_std": 0.2906969487667084} {"timestamp_utc": "2026-04-12T23:28:17Z", "mode": "train", "global_step": 514, "epoch": 0.05163234555499749, "loss": 0.0554, "grad_norm": 15.705352783203125, "learning_rate": 8.445454545454547e-06, "num_tokens": 974996.0, "completions/mean_length": 27.75, "completions/min_length": 23.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.75, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.7156796455383301, "rewards/meter/std": 0.2974271774291992, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9971784353256226, "rewards/repeat_soft/std": 0.005066921003162861, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.6988986730575562, "rewards/total_composite/std": 0.1326158344745636, "reward": 0.6988986730575562, "reward_std": 0.1326158344745636, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16823437809944153, "sampling/sampling_logp_difference/max": 1.501335620880127, "sampling/importance_sampling_ratio/min": 0.2228323519229889, "sampling/importance_sampling_ratio/mean": 1.04860520362854, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1484571695327759, "clip_ratio/low_mean": 0.054549697786569595, "clip_ratio/low_min": 0.054549697786569595, "clip_ratio/high_mean": 0.10542206559330225, "clip_ratio/high_max": 0.10542206559330225, "clip_ratio/region_mean": 0.15997176337987185, "reward_total_mean": 0.6988986730575562, "reward_meter_mean": 0.7156796455383301, "reward_meter_std": 0.2974271774291992, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9971784353256226, "reward_repeat_soft_std": 0.005066921003162861, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.6988986730575562, "reward_total_composite_std": 0.1326158344745636} {"timestamp_utc": "2026-04-12T23:28:23Z", "mode": "train", "global_step": 515, "epoch": 0.05173279758915118, "loss": 0.1162, "grad_norm": 23.387174606323242, "learning_rate": 8.442424242424243e-06, "num_tokens": 976438.0, "completions/mean_length": 29.25, "completions/min_length": 22.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.25, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.5186865329742432, "rewards/meter/std": 0.36112794280052185, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9969335794448853, "rewards/repeat_soft/std": 0.005040564574301243, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.6203523278236389, "rewards/total_composite/std": 0.1588342934846878, "reward": 0.6203523278236389, "reward_std": 0.1588342934846878, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21727138757705688, "sampling/sampling_logp_difference/max": 1.7260942459106445, "sampling/importance_sampling_ratio/min": 0.17797818779945374, "sampling/importance_sampling_ratio/mean": 1.058390498161316, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9467035084962845, "clip_ratio/low_mean": 0.14786527305841446, "clip_ratio/low_min": 0.14786527305841446, "clip_ratio/high_mean": 0.10730146616697311, "clip_ratio/high_max": 0.10730146616697311, "clip_ratio/region_mean": 0.2551667392253876, "reward_total_mean": 0.6203523278236389, "reward_meter_mean": 0.5186865329742432, "reward_meter_std": 0.36112794280052185, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9969335794448853, "reward_repeat_soft_std": 0.005040564574301243, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.6203523278236389, "reward_total_composite_std": 0.1588342934846878} {"timestamp_utc": "2026-04-12T23:28:29Z", "mode": "train", "global_step": 516, "epoch": 0.05183324962330487, "loss": 0.0312, "grad_norm": 15.843986511230469, "learning_rate": 8.43939393939394e-06, "num_tokens": 978202.0, "completions/mean_length": 54.5, "completions/min_length": 41.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.5, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9018036127090454, "rewards/meter/std": 0.14877888560295105, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9329791069030762, "rewards/repeat_soft/std": 0.09757678955793381, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.10392305999994278, "rewards/total_composite/mean": 0.788609504699707, "rewards/total_composite/std": 0.07848219573497772, "reward": 0.788609504699707, "reward_std": 0.07848219573497772, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16406294703483582, "sampling/sampling_logp_difference/max": 1.597320556640625, "sampling/importance_sampling_ratio/min": 0.20243822038173676, "sampling/importance_sampling_ratio/mean": 1.0343953371047974, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4081602692604065, "clip_ratio/low_mean": 0.056880729272961617, "clip_ratio/low_min": 0.056880729272961617, "clip_ratio/high_mean": 0.09729034919291735, "clip_ratio/high_max": 0.09729034919291735, "clip_ratio/region_mean": 0.15417107846587896, "reward_total_mean": 0.788609504699707, "reward_meter_mean": 0.9018036127090454, "reward_meter_std": 0.14877888560295105, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9329791069030762, "reward_repeat_soft_std": 0.09757678955793381, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.10392305999994278, "reward_total_composite_mean": 0.788609504699707, "reward_total_composite_std": 0.07848219573497772} {"timestamp_utc": "2026-04-12T23:28:37Z", "mode": "train", "global_step": 517, "epoch": 0.051933701657458566, "loss": 0.186, "grad_norm": 13.771576881408691, "learning_rate": 8.436363636363637e-06, "num_tokens": 980028.0, "completions/mean_length": 68.25, "completions/min_length": 34.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.25, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.43021270632743835, "rewards/meter/std": 0.3466002941131592, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9921597838401794, "rewards/repeat_soft/std": 0.009790941141545773, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.5202833414077759, "rewards/total_composite/std": 0.2692619860172272, "reward": 0.5202833414077759, "reward_std": 0.2692619860172272, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2190454751253128, "sampling/sampling_logp_difference/max": 1.6765108108520508, "sampling/importance_sampling_ratio/min": 0.18702539801597595, "sampling/importance_sampling_ratio/mean": 1.024280071258545, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4004815816879272, "clip_ratio/low_mean": 0.07049572467803955, "clip_ratio/low_min": 0.07049572467803955, "clip_ratio/high_mean": 0.12294801324605942, "clip_ratio/high_max": 0.12294801324605942, "clip_ratio/region_mean": 0.19344373792409897, "reward_total_mean": 0.5202833414077759, "reward_meter_mean": 0.43021270632743835, "reward_meter_std": 0.3466002941131592, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9921597838401794, "reward_repeat_soft_std": 0.009790941141545773, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.5202833414077759, "reward_total_composite_std": 0.2692619860172272} {"timestamp_utc": "2026-04-12T23:28:48Z", "mode": "train", "global_step": 518, "epoch": 0.05203415369161225, "loss": -0.0858, "grad_norm": 5.074728012084961, "learning_rate": 8.433333333333334e-06, "num_tokens": 981921.0, "completions/mean_length": 132.625, "completions/min_length": 69.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 78.42857360839844, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.601528525352478, "rewards/meter/std": 0.38813698291778564, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9885894060134888, "rewards/repeat_soft/std": 0.011425790376961231, "rewards/judge_quality/mean": 0.4112499952316284, "rewards/judge_quality/std": 0.1797965168952942, "rewards/total_composite/mean": 0.5060954093933105, "rewards/total_composite/std": 0.3489855229854584, "reward": 0.5060954093933105, "reward_std": 0.3489855229854584, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18346500396728516, "sampling/sampling_logp_difference/max": 2.252902030944824, "sampling/importance_sampling_ratio/min": 0.10509379953145981, "sampling/importance_sampling_ratio/mean": 1.0463347434997559, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8208155632019043, "clip_ratio/low_mean": 0.06321985088288784, "clip_ratio/low_min": 0.06321985088288784, "clip_ratio/high_mean": 0.07525924313813448, "clip_ratio/high_max": 0.07525924313813448, "clip_ratio/region_mean": 0.13847909402102232, "reward_total_mean": 0.5060954093933105, "reward_meter_mean": 0.601528525352478, "reward_meter_std": 0.38813698291778564, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9885894060134888, "reward_repeat_soft_std": 0.011425790376961231, "reward_judge_quality_mean": 0.4112499952316284, "reward_judge_quality_std": 0.1797965168952942, "reward_total_composite_mean": 0.5060954093933105, "reward_total_composite_std": 0.3489855229854584} {"timestamp_utc": "2026-04-12T23:28:58Z", "mode": "train", "global_step": 519, "epoch": 0.05213460572576595, "loss": -0.0497, "grad_norm": 20.88576889038086, "learning_rate": 8.43030303030303e-06, "num_tokens": 983512.0, "completions/mean_length": 22.875, "completions/min_length": 18.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.875, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.567440390586853, "rewards/meter/std": 0.4502256512641907, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.5674999952316284, "rewards/judge_quality/std": 0.21756774187088013, "rewards/total_composite/mean": 0.6718481779098511, "rewards/total_composite/std": 0.18193823099136353, "reward": 0.6718481779098511, "reward_std": 0.18193824589252472, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19241707026958466, "sampling/sampling_logp_difference/max": 1.3387165069580078, "sampling/importance_sampling_ratio/min": 0.26218193769454956, "sampling/importance_sampling_ratio/mean": 1.015006422996521, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8213015496730804, "clip_ratio/low_mean": 0.10517098801210523, "clip_ratio/low_min": 0.10517098801210523, "clip_ratio/high_mean": 0.1023214291781187, "clip_ratio/high_max": 0.1023214291781187, "clip_ratio/region_mean": 0.20749241719022393, "reward_total_mean": 0.6718481779098511, "reward_meter_mean": 0.567440390586853, "reward_meter_std": 0.4502256512641907, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.5674999952316284, "reward_judge_quality_std": 0.21756774187088013, "reward_total_composite_mean": 0.6718481779098511, "reward_total_composite_std": 0.18193823099136353} {"timestamp_utc": "2026-04-12T23:29:04Z", "mode": "train", "global_step": 520, "epoch": 0.052235057759919636, "loss": 0.0412, "grad_norm": 21.046466827392578, "learning_rate": 8.427272727272729e-06, "num_tokens": 984971.0, "completions/mean_length": 30.375, "completions/min_length": 28.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.375, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.3689895272254944, "rewards/meter/std": 0.3590598702430725, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.998379647731781, "rewards/repeat_soft/std": 0.003917992115020752, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.5400082468986511, "rewards/total_composite/std": 0.16603009402751923, "reward": 0.5400082468986511, "reward_std": 0.16603007912635803, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15813952684402466, "sampling/sampling_logp_difference/max": 1.8285198211669922, "sampling/importance_sampling_ratio/min": 0.17918173968791962, "sampling/importance_sampling_ratio/mean": 1.0426687002182007, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9964261576533318, "clip_ratio/low_mean": 0.08722589071840048, "clip_ratio/low_min": 0.08722589071840048, "clip_ratio/high_mean": 0.059770116582512856, "clip_ratio/high_max": 0.059770116582512856, "clip_ratio/region_mean": 0.14699600730091333, "reward_total_mean": 0.5400082468986511, "reward_meter_mean": 0.3689895272254944, "reward_meter_std": 0.3590598702430725, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.998379647731781, "reward_repeat_soft_std": 0.003917992115020752, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.5400082468986511, "reward_total_composite_std": 0.16603009402751923} {"timestamp_utc": "2026-04-12T23:29:12Z", "mode": "train", "global_step": 521, "epoch": 0.05233550979407333, "loss": 0.0193, "grad_norm": 10.962797164916992, "learning_rate": 8.424242424242425e-06, "num_tokens": 986995.0, "completions/mean_length": 75.0, "completions/min_length": 62.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.4840908944606781, "rewards/meter/std": 0.3235366940498352, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9778538942337036, "rewards/repeat_soft/std": 0.04227091744542122, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.2030482292175293, "rewards/total_composite/mean": 0.46472644805908203, "rewards/total_composite/std": 0.33120009303092957, "reward": 0.46472644805908203, "reward_std": 0.3312000632286072, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2124103605747223, "sampling/sampling_logp_difference/max": 1.4307355880737305, "sampling/importance_sampling_ratio/min": 0.23913295567035675, "sampling/importance_sampling_ratio/mean": 1.058887004852295, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6381392776966095, "clip_ratio/low_mean": 0.0646372139453888, "clip_ratio/low_min": 0.0646372139453888, "clip_ratio/high_mean": 0.09469629637897015, "clip_ratio/high_max": 0.09469629637897015, "clip_ratio/region_mean": 0.15933351032435894, "reward_total_mean": 0.46472644805908203, "reward_meter_mean": 0.4840908944606781, "reward_meter_std": 0.3235366940498352, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9778538942337036, "reward_repeat_soft_std": 0.04227091744542122, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.2030482292175293, "reward_total_composite_mean": 0.46472644805908203, "reward_total_composite_std": 0.33120009303092957} {"timestamp_utc": "2026-04-12T23:29:18Z", "mode": "train", "global_step": 522, "epoch": 0.052435961828227025, "loss": 0.0501, "grad_norm": 13.829574584960938, "learning_rate": 8.421212121212122e-06, "num_tokens": 988745.0, "completions/mean_length": 56.75, "completions/min_length": 50.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.75, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.45861226320266724, "rewards/meter/std": 0.38577523827552795, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9918134808540344, "rewards/repeat_soft/std": 0.006101328879594803, "rewards/judge_quality/mean": 0.5024999976158142, "rewards/judge_quality/std": 0.1348809152841568, "rewards/total_composite/mean": 0.6063068509101868, "rewards/total_composite/std": 0.18873342871665955, "reward": 0.6063068509101868, "reward_std": 0.18873342871665955, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21301758289337158, "sampling/sampling_logp_difference/max": 1.4112606048583984, "sampling/importance_sampling_ratio/min": 0.2438357025384903, "sampling/importance_sampling_ratio/mean": 1.0601171255111694, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.58821140229702, "clip_ratio/low_mean": 0.08224456384778023, "clip_ratio/low_min": 0.08224456384778023, "clip_ratio/high_mean": 0.09144110977649689, "clip_ratio/high_max": 0.09144110977649689, "clip_ratio/region_mean": 0.17368567362427711, "reward_total_mean": 0.6063068509101868, "reward_meter_mean": 0.45861226320266724, "reward_meter_std": 0.38577523827552795, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9918134808540344, "reward_repeat_soft_std": 0.006101328879594803, "reward_judge_quality_mean": 0.5024999976158142, "reward_judge_quality_std": 0.1348809152841568, "reward_total_composite_mean": 0.6063068509101868, "reward_total_composite_std": 0.18873342871665955} {"timestamp_utc": "2026-04-12T23:29:25Z", "mode": "train", "global_step": 523, "epoch": 0.05253641386238071, "loss": -0.0029, "grad_norm": 15.10610294342041, "learning_rate": 8.418181818181819e-06, "num_tokens": 990542.0, "completions/mean_length": 59.625, "completions/min_length": 53.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.625, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.41004711389541626, "rewards/meter/std": 0.23012922704219818, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.997488260269165, "rewards/repeat_soft/std": 0.0036881056148558855, "rewards/judge_quality/mean": 0.5974999666213989, "rewards/judge_quality/std": 0.2499571591615677, "rewards/total_composite/mean": 0.6135200262069702, "rewards/total_composite/std": 0.14218947291374207, "reward": 0.6135200262069702, "reward_std": 0.14218944311141968, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19524744153022766, "sampling/sampling_logp_difference/max": 2.1830172538757324, "sampling/importance_sampling_ratio/min": 0.11270096898078918, "sampling/importance_sampling_ratio/mean": 1.0263546705245972, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.132021389901638, "clip_ratio/low_mean": 0.06563308089971542, "clip_ratio/low_min": 0.06563308089971542, "clip_ratio/high_mean": 0.10940296202898026, "clip_ratio/high_max": 0.10940296202898026, "clip_ratio/region_mean": 0.17503604292869568, "reward_total_mean": 0.6135200262069702, "reward_meter_mean": 0.41004711389541626, "reward_meter_std": 0.23012922704219818, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.997488260269165, "reward_repeat_soft_std": 0.0036881056148558855, "reward_judge_quality_mean": 0.5974999666213989, "reward_judge_quality_std": 0.2499571591615677, "reward_total_composite_mean": 0.6135200262069702, "reward_total_composite_std": 0.14218947291374207} {"timestamp_utc": "2026-04-12T23:29:31Z", "mode": "train", "global_step": 524, "epoch": 0.05263686589653441, "loss": -0.0316, "grad_norm": 19.398305892944336, "learning_rate": 8.415151515151516e-06, "num_tokens": 991954.0, "completions/mean_length": 25.5, "completions/min_length": 23.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.5, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.4611278772354126, "rewards/meter/std": 0.28069302439689636, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.989084005355835, "rewards/repeat_soft/std": 0.00421832175925374, "rewards/judge_quality/mean": 0.9237500429153442, "rewards/judge_quality/std": 0.01060659158974886, "rewards/total_composite/mean": 0.7335410118103027, "rewards/total_composite/std": 0.1256505250930786, "reward": 0.7335410118103027, "reward_std": 0.1256505250930786, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10641678422689438, "sampling/sampling_logp_difference/max": 1.1154727935791016, "sampling/importance_sampling_ratio/min": 0.3277602791786194, "sampling/importance_sampling_ratio/mean": 0.9993792176246643, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5373994037508965, "clip_ratio/low_mean": 0.055064103566110134, "clip_ratio/low_min": 0.055064103566110134, "clip_ratio/high_mean": 0.07369376253336668, "clip_ratio/high_max": 0.07369376253336668, "clip_ratio/region_mean": 0.12875786609947681, "reward_total_mean": 0.7335410118103027, "reward_meter_mean": 0.4611278772354126, "reward_meter_std": 0.28069302439689636, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.989084005355835, "reward_repeat_soft_std": 0.00421832175925374, "reward_judge_quality_mean": 0.9237500429153442, "reward_judge_quality_std": 0.01060659158974886, "reward_total_composite_mean": 0.7335410118103027, "reward_total_composite_std": 0.1256505250930786} {"timestamp_utc": "2026-04-12T23:29:42Z", "mode": "train", "global_step": 525, "epoch": 0.052737317930688095, "loss": -0.0817, "grad_norm": 4.123688220977783, "learning_rate": 8.412121212121212e-06, "num_tokens": 993534.0, "completions/mean_length": 99.5, "completions/min_length": 28.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 40.57143020629883, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.5521451830863953, "rewards/meter/std": 0.37825945019721985, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9997996091842651, "rewards/repeat_soft/std": 0.0005668329540640116, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.13617216050624847, "rewards/total_composite/mean": 0.56163489818573, "rewards/total_composite/std": 0.2814488410949707, "reward": 0.56163489818573, "reward_std": 0.2814488410949707, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23055578768253326, "sampling/sampling_logp_difference/max": 1.804037094116211, "sampling/importance_sampling_ratio/min": 0.1646329015493393, "sampling/importance_sampling_ratio/mean": 1.082234263420105, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4051999747753143, "clip_ratio/low_mean": 0.09532608650624752, "clip_ratio/low_min": 0.09532608650624752, "clip_ratio/high_mean": 0.09493224136531353, "clip_ratio/high_max": 0.09493224136531353, "clip_ratio/region_mean": 0.19025832787156105, "reward_total_mean": 0.56163489818573, "reward_meter_mean": 0.5521451830863953, "reward_meter_std": 0.37825945019721985, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9997996091842651, "reward_repeat_soft_std": 0.0005668329540640116, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.13617216050624847, "reward_total_composite_mean": 0.56163489818573, "reward_total_composite_std": 0.2814488410949707} {"timestamp_utc": "2026-04-12T23:29:48Z", "mode": "train", "global_step": 526, "epoch": 0.05283776996484179, "loss": 0.1079, "grad_norm": 26.08743667602539, "learning_rate": 8.40909090909091e-06, "num_tokens": 995027.0, "completions/mean_length": 28.625, "completions/min_length": 24.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.625, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.38713985681533813, "rewards/meter/std": 0.3541550040245056, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9942673444747925, "rewards/repeat_soft/std": 0.008106804452836514, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.1011011004447937, "rewards/total_composite/mean": 0.5653896927833557, "rewards/total_composite/std": 0.154084712266922, "reward": 0.5653896927833557, "reward_std": 0.154084712266922, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1857023984193802, "sampling/sampling_logp_difference/max": 1.2012571096420288, "sampling/importance_sampling_ratio/min": 0.3008158206939697, "sampling/importance_sampling_ratio/mean": 1.0323069095611572, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.07775117456913, "clip_ratio/low_mean": 0.09358360897749662, "clip_ratio/low_min": 0.09358360897749662, "clip_ratio/high_mean": 0.07239494239911437, "clip_ratio/high_max": 0.07239494239911437, "clip_ratio/region_mean": 0.165978551376611, "reward_total_mean": 0.5653896927833557, "reward_meter_mean": 0.38713985681533813, "reward_meter_std": 0.3541550040245056, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9942673444747925, "reward_repeat_soft_std": 0.008106804452836514, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.1011011004447937, "reward_total_composite_mean": 0.5653896927833557, "reward_total_composite_std": 0.154084712266922} {"timestamp_utc": "2026-04-12T23:29:54Z", "mode": "train", "global_step": 527, "epoch": 0.05293822199899548, "loss": 0.1106, "grad_norm": 12.550690650939941, "learning_rate": 8.406060606060606e-06, "num_tokens": 996773.0, "completions/mean_length": 56.25, "completions/min_length": 48.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.25, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.27778464555740356, "rewards/meter/std": 0.22641783952713013, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9944825172424316, "rewards/repeat_soft/std": 0.004231174476444721, "rewards/judge_quality/mean": 0.5487499833106995, "rewards/judge_quality/std": 0.1724974364042282, "rewards/total_composite/mean": 0.5343887805938721, "rewards/total_composite/std": 0.125327005982399, "reward": 0.5343887805938721, "reward_std": 0.125327005982399, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16170598566532135, "sampling/sampling_logp_difference/max": 1.4458556175231934, "sampling/importance_sampling_ratio/min": 0.2809827923774719, "sampling/importance_sampling_ratio/mean": 1.033893346786499, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4598309621214867, "clip_ratio/low_mean": 0.09054881148040295, "clip_ratio/low_min": 0.09054881148040295, "clip_ratio/high_mean": 0.11094221472740173, "clip_ratio/high_max": 0.11094221472740173, "clip_ratio/region_mean": 0.20149102620780468, "reward_total_mean": 0.5343887805938721, "reward_meter_mean": 0.27778464555740356, "reward_meter_std": 0.22641783952713013, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9944825172424316, "reward_repeat_soft_std": 0.004231174476444721, "reward_judge_quality_mean": 0.5487499833106995, "reward_judge_quality_std": 0.1724974364042282, "reward_total_composite_mean": 0.5343887805938721, "reward_total_composite_std": 0.125327005982399} {"timestamp_utc": "2026-04-12T23:30:00Z", "mode": "train", "global_step": 528, "epoch": 0.05303867403314917, "loss": -0.0074, "grad_norm": 13.064423561096191, "learning_rate": 8.403030303030304e-06, "num_tokens": 998454.0, "completions/mean_length": 52.125, "completions/min_length": 45.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.125, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.7463736534118652, "rewards/meter/std": 0.3120080530643463, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9914469718933105, "rewards/repeat_soft/std": 0.007205803878605366, "rewards/judge_quality/mean": 0.7062500715255737, "rewards/judge_quality/std": 0.18638958036899567, "rewards/total_composite/mean": 0.7968878149986267, "rewards/total_composite/std": 0.1636294424533844, "reward": 0.7968878149986267, "reward_std": 0.1636294275522232, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18977224826812744, "sampling/sampling_logp_difference/max": 3.4132673740386963, "sampling/importance_sampling_ratio/min": 0.032933417707681656, "sampling/importance_sampling_ratio/mean": 1.0449668169021606, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6219849288463593, "clip_ratio/low_mean": 0.07251915708184242, "clip_ratio/low_min": 0.07251915708184242, "clip_ratio/high_mean": 0.1128940749913454, "clip_ratio/high_max": 0.1128940749913454, "clip_ratio/region_mean": 0.18541323207318783, "reward_total_mean": 0.7968878149986267, "reward_meter_mean": 0.7463736534118652, "reward_meter_std": 0.3120080530643463, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9914469718933105, "reward_repeat_soft_std": 0.007205803878605366, "reward_judge_quality_mean": 0.7062500715255737, "reward_judge_quality_std": 0.18638958036899567, "reward_total_composite_mean": 0.7968878149986267, "reward_total_composite_std": 0.1636294424533844} {"timestamp_utc": "2026-04-12T23:30:06Z", "mode": "train", "global_step": 529, "epoch": 0.05313912606730286, "loss": 0.0317, "grad_norm": 16.8637638092041, "learning_rate": 8.400000000000001e-06, "num_tokens": 1000071.0, "completions/mean_length": 32.125, "completions/min_length": 29.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.125, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.2584942579269409, "rewards/meter/std": 0.2840186655521393, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9957189559936523, "rewards/repeat_soft/std": 0.005848255939781666, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.5163060426712036, "rewards/total_composite/std": 0.2486552894115448, "reward": 0.5163060426712036, "reward_std": 0.2486552745103836, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17351149022579193, "sampling/sampling_logp_difference/max": 1.3705486059188843, "sampling/importance_sampling_ratio/min": 0.2539675831794739, "sampling/importance_sampling_ratio/mean": 1.0470540523529053, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3388183936476707, "clip_ratio/low_mean": 0.06061422452330589, "clip_ratio/low_min": 0.06061422452330589, "clip_ratio/high_mean": 0.13391490932554007, "clip_ratio/high_max": 0.13391490932554007, "clip_ratio/region_mean": 0.19452913384884596, "reward_total_mean": 0.5163060426712036, "reward_meter_mean": 0.2584942579269409, "reward_meter_std": 0.2840186655521393, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9957189559936523, "reward_repeat_soft_std": 0.005848255939781666, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.5163060426712036, "reward_total_composite_std": 0.2486552894115448} {"timestamp_utc": "2026-04-12T23:30:13Z", "mode": "train", "global_step": 530, "epoch": 0.053239578101456554, "loss": 0.0305, "grad_norm": 9.083611488342285, "learning_rate": 8.396969696969698e-06, "num_tokens": 1001936.0, "completions/mean_length": 65.125, "completions/min_length": 61.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.125, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.5919568538665771, "rewards/meter/std": 0.28676894307136536, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9978128671646118, "rewards/repeat_soft/std": 0.0030486651230603456, "rewards/judge_quality/mean": 0.5774999856948853, "rewards/judge_quality/std": 0.1527603417634964, "rewards/total_composite/mean": 0.6894118785858154, "rewards/total_composite/std": 0.12096500396728516, "reward": 0.6894118785858154, "reward_std": 0.12096499651670456, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16135084629058838, "sampling/sampling_logp_difference/max": 1.1142768859863281, "sampling/importance_sampling_ratio/min": 0.32815247774124146, "sampling/importance_sampling_ratio/mean": 1.048598051071167, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7723905146121979, "clip_ratio/low_mean": 0.05821434408426285, "clip_ratio/low_min": 0.05821434408426285, "clip_ratio/high_mean": 0.07409198954701424, "clip_ratio/high_max": 0.07409198954701424, "clip_ratio/region_mean": 0.13230633363127708, "reward_total_mean": 0.6894118785858154, "reward_meter_mean": 0.5919568538665771, "reward_meter_std": 0.28676894307136536, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9978128671646118, "reward_repeat_soft_std": 0.0030486651230603456, "reward_judge_quality_mean": 0.5774999856948853, "reward_judge_quality_std": 0.1527603417634964, "reward_total_composite_mean": 0.6894118785858154, "reward_total_composite_std": 0.12096500396728516} {"timestamp_utc": "2026-04-12T23:30:20Z", "mode": "train", "global_step": 531, "epoch": 0.05334003013561025, "loss": -0.0032, "grad_norm": 19.016307830810547, "learning_rate": 8.393939393939394e-06, "num_tokens": 1003446.0, "completions/mean_length": 37.75, "completions/min_length": 28.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.75, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.44900816679000854, "rewards/meter/std": 0.3307282030582428, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9888160824775696, "rewards/repeat_soft/std": 0.011213903315365314, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.6019352674484253, "rewards/total_composite/std": 0.13139669597148895, "reward": 0.6019352674484253, "reward_std": 0.13139669597148895, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1717928797006607, "sampling/sampling_logp_difference/max": 1.2405962944030762, "sampling/importance_sampling_ratio/min": 0.296209454536438, "sampling/importance_sampling_ratio/mean": 0.9991665482521057, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4964423105120659, "clip_ratio/low_mean": 0.08391383104026318, "clip_ratio/low_min": 0.08391383104026318, "clip_ratio/high_mean": 0.08119402453303337, "clip_ratio/high_max": 0.08119402453303337, "clip_ratio/region_mean": 0.16510785557329655, "reward_total_mean": 0.6019352674484253, "reward_meter_mean": 0.44900816679000854, "reward_meter_std": 0.3307282030582428, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9888160824775696, "reward_repeat_soft_std": 0.011213903315365314, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.6019352674484253, "reward_total_composite_std": 0.13139669597148895} {"timestamp_utc": "2026-04-12T23:30:26Z", "mode": "train", "global_step": 532, "epoch": 0.053440482169763937, "loss": 0.0489, "grad_norm": 11.768227577209473, "learning_rate": 8.390909090909091e-06, "num_tokens": 1005103.0, "completions/mean_length": 41.125, "completions/min_length": 39.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.125, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.8119117021560669, "rewards/meter/std": 0.3484020531177521, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.947367787361145, "rewards/repeat_soft/std": 0.04202844947576523, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7372220754623413, "rewards/total_composite/std": 0.15388886630535126, "reward": 0.7372220754623413, "reward_std": 0.15388885140419006, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17298851907253265, "sampling/sampling_logp_difference/max": 1.25689697265625, "sampling/importance_sampling_ratio/min": 0.28453558683395386, "sampling/importance_sampling_ratio/mean": 1.0268117189407349, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5163950473070145, "clip_ratio/low_mean": 0.054166668094694614, "clip_ratio/low_min": 0.054166668094694614, "clip_ratio/high_mean": 0.10632533766329288, "clip_ratio/high_max": 0.10632533766329288, "clip_ratio/region_mean": 0.1604920057579875, "reward_total_mean": 0.7372220754623413, "reward_meter_mean": 0.8119117021560669, "reward_meter_std": 0.3484020531177521, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.947367787361145, "reward_repeat_soft_std": 0.04202844947576523, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7372220754623413, "reward_total_composite_std": 0.15388886630535126} {"timestamp_utc": "2026-04-12T23:30:32Z", "mode": "train", "global_step": 533, "epoch": 0.05354093420391763, "loss": -0.0342, "grad_norm": 14.499860763549805, "learning_rate": 8.387878787878788e-06, "num_tokens": 1006585.0, "completions/mean_length": 32.25, "completions/min_length": 27.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.25, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.6116003394126892, "rewards/meter/std": 0.3941424787044525, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.989039957523346, "rewards/repeat_soft/std": 0.009066439233720303, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.6801241636276245, "rewards/total_composite/std": 0.21325655281543732, "reward": 0.6801241636276245, "reward_std": 0.21325653791427612, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18959642946720123, "sampling/sampling_logp_difference/max": 1.3138618469238281, "sampling/importance_sampling_ratio/min": 0.26878008246421814, "sampling/importance_sampling_ratio/mean": 1.0507913827896118, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9196448624134064, "clip_ratio/low_mean": 0.10428240709006786, "clip_ratio/low_min": 0.10428240709006786, "clip_ratio/high_mean": 0.08911602851003408, "clip_ratio/high_max": 0.08911602851003408, "clip_ratio/region_mean": 0.19339843560010195, "reward_total_mean": 0.6801241636276245, "reward_meter_mean": 0.6116003394126892, "reward_meter_std": 0.3941424787044525, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.989039957523346, "reward_repeat_soft_std": 0.009066439233720303, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.6801241636276245, "reward_total_composite_std": 0.21325655281543732} {"timestamp_utc": "2026-04-12T23:30:38Z", "mode": "train", "global_step": 534, "epoch": 0.05364138623807132, "loss": 0.0533, "grad_norm": 24.730199813842773, "learning_rate": 8.384848484848485e-06, "num_tokens": 1008146.0, "completions/mean_length": 21.125, "completions/min_length": 16.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.125, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.7965233325958252, "rewards/meter/std": 0.25607559084892273, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.19334925711154938, "rewards/total_composite/mean": 0.7453104853630066, "rewards/total_composite/std": 0.1324642449617386, "reward": 0.7453104853630066, "reward_std": 0.1324642151594162, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21982233226299286, "sampling/sampling_logp_difference/max": 1.4278974533081055, "sampling/importance_sampling_ratio/min": 0.23981261253356934, "sampling/importance_sampling_ratio/mean": 1.0200767517089844, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5316249281167984, "clip_ratio/low_mean": 0.059780498035252094, "clip_ratio/low_min": 0.059780498035252094, "clip_ratio/high_mean": 0.0939741269685328, "clip_ratio/high_max": 0.0939741269685328, "clip_ratio/region_mean": 0.1537546250037849, "reward_total_mean": 0.7453104853630066, "reward_meter_mean": 0.7965233325958252, "reward_meter_std": 0.25607559084892273, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.19334925711154938, "reward_total_composite_mean": 0.7453104853630066, "reward_total_composite_std": 0.1324642449617386} {"timestamp_utc": "2026-04-12T23:30:50Z", "mode": "train", "global_step": 535, "epoch": 0.053741838272225013, "loss": -0.1461, "grad_norm": 2.6066408157348633, "learning_rate": 8.381818181818183e-06, "num_tokens": 1010072.0, "completions/mean_length": 193.75, "completions/min_length": 61.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 87.66667175292969, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.6999333500862122, "rewards/meter/std": 0.38244709372520447, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9965536594390869, "rewards/repeat_soft/std": 0.006309994030743837, "rewards/judge_quality/mean": 0.3062499761581421, "rewards/judge_quality/std": 0.1686871498823166, "rewards/total_composite/mean": 0.5074477195739746, "rewards/total_composite/std": 0.3550569415092468, "reward": 0.5074477195739746, "reward_std": 0.35505691170692444, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2091653048992157, "sampling/sampling_logp_difference/max": 1.475672721862793, "sampling/importance_sampling_ratio/min": 0.2286248803138733, "sampling/importance_sampling_ratio/mean": 1.0493814945220947, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4933042526245117, "clip_ratio/low_mean": 0.01659291982650757, "clip_ratio/low_min": 0.01659291982650757, "clip_ratio/high_mean": 0.1186831509694457, "clip_ratio/high_max": 0.1186831509694457, "clip_ratio/region_mean": 0.13527607079595327, "reward_total_mean": 0.5074477195739746, "reward_meter_mean": 0.6999333500862122, "reward_meter_std": 0.38244709372520447, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9965536594390869, "reward_repeat_soft_std": 0.006309994030743837, "reward_judge_quality_mean": 0.3062499761581421, "reward_judge_quality_std": 0.1686871498823166, "reward_total_composite_mean": 0.5074477195739746, "reward_total_composite_std": 0.3550569415092468} {"timestamp_utc": "2026-04-12T23:30:57Z", "mode": "train", "global_step": 536, "epoch": 0.0538422903063787, "loss": 0.0612, "grad_norm": 13.26952838897705, "learning_rate": 8.37878787878788e-06, "num_tokens": 1012132.0, "completions/mean_length": 76.5, "completions/min_length": 66.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.5, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.5800155401229858, "rewards/meter/std": 0.2211722582578659, "rewards/count_adherence/mean": 0.9166666269302368, "rewards/count_adherence/std": 0.0890870913863182, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9895872473716736, "rewards/repeat_soft/std": 0.005979498848319054, "rewards/judge_quality/mean": 0.5699999928474426, "rewards/judge_quality/std": 0.16035676002502441, "rewards/total_composite/mean": 0.6684657335281372, "rewards/total_composite/std": 0.09108790755271912, "reward": 0.6684657335281372, "reward_std": 0.09108790010213852, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1496545970439911, "sampling/sampling_logp_difference/max": 1.5831339359283447, "sampling/importance_sampling_ratio/min": 0.20533059537410736, "sampling/importance_sampling_ratio/mean": 0.9993453621864319, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8538849875330925, "clip_ratio/low_mean": 0.02356135006994009, "clip_ratio/low_min": 0.02356135006994009, "clip_ratio/high_mean": 0.12080173566937447, "clip_ratio/high_max": 0.12080173566937447, "clip_ratio/region_mean": 0.14436308573931456, "reward_total_mean": 0.6684657335281372, "reward_meter_mean": 0.5800155401229858, "reward_meter_std": 0.2211722582578659, "reward_count_adherence_mean": 0.9166666269302368, "reward_count_adherence_std": 0.0890870913863182, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9895872473716736, "reward_repeat_soft_std": 0.005979498848319054, "reward_judge_quality_mean": 0.5699999928474426, "reward_judge_quality_std": 0.16035676002502441, "reward_total_composite_mean": 0.6684657335281372, "reward_total_composite_std": 0.09108790755271912} {"timestamp_utc": "2026-04-12T23:31:03Z", "mode": "train", "global_step": 537, "epoch": 0.053942742340532396, "loss": -0.0123, "grad_norm": 18.393840789794922, "learning_rate": 8.375757575757576e-06, "num_tokens": 1013716.0, "completions/mean_length": 27.0, "completions/min_length": 22.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.0, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.48866334557533264, "rewards/meter/std": 0.39294004440307617, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9729256629943848, "rewards/repeat_soft/std": 0.06289859116077423, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.6869410276412964, "rewards/total_composite/std": 0.19635282456874847, "reward": 0.6869410276412964, "reward_std": 0.19635282456874847, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17208632826805115, "sampling/sampling_logp_difference/max": 2.2594428062438965, "sampling/importance_sampling_ratio/min": 0.1044086441397667, "sampling/importance_sampling_ratio/mean": 1.0277398824691772, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.169266790151596, "clip_ratio/low_mean": 0.11721403803676367, "clip_ratio/low_min": 0.11721403803676367, "clip_ratio/high_mean": 0.07093735318630934, "clip_ratio/high_max": 0.07093735318630934, "clip_ratio/region_mean": 0.188151391223073, "reward_total_mean": 0.6869410276412964, "reward_meter_mean": 0.48866334557533264, "reward_meter_std": 0.39294004440307617, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9729256629943848, "reward_repeat_soft_std": 0.06289859116077423, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.6869410276412964, "reward_total_composite_std": 0.19635282456874847} {"timestamp_utc": "2026-04-12T23:31:14Z", "mode": "train", "global_step": 538, "epoch": 0.05404319437468609, "loss": -0.1095, "grad_norm": 3.79415225982666, "learning_rate": 8.372727272727273e-06, "num_tokens": 1015519.0, "completions/mean_length": 112.375, "completions/min_length": 48.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 55.28571701049805, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.5696316361427307, "rewards/meter/std": 0.4400528073310852, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9972052574157715, "rewards/repeat_soft/std": 0.003159511834383011, "rewards/judge_quality/mean": 0.5737500190734863, "rewards/judge_quality/std": 0.30514341592788696, "rewards/total_composite/mean": 0.609243631362915, "rewards/total_composite/std": 0.30942168831825256, "reward": 0.609243631362915, "reward_std": 0.30942171812057495, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1884312778711319, "sampling/sampling_logp_difference/max": 1.6492009162902832, "sampling/importance_sampling_ratio/min": 0.19220343232154846, "sampling/importance_sampling_ratio/mean": 1.032251000404358, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5039555430412292, "clip_ratio/low_mean": 0.06377100013196468, "clip_ratio/low_min": 0.06377100013196468, "clip_ratio/high_mean": 0.09418723545968533, "clip_ratio/high_max": 0.09418723545968533, "clip_ratio/region_mean": 0.15795823559165, "reward_total_mean": 0.609243631362915, "reward_meter_mean": 0.5696316361427307, "reward_meter_std": 0.4400528073310852, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9972052574157715, "reward_repeat_soft_std": 0.003159511834383011, "reward_judge_quality_mean": 0.5737500190734863, "reward_judge_quality_std": 0.30514341592788696, "reward_total_composite_mean": 0.609243631362915, "reward_total_composite_std": 0.30942168831825256} {"timestamp_utc": "2026-04-12T23:31:21Z", "mode": "train", "global_step": 539, "epoch": 0.05414364640883978, "loss": 0.0486, "grad_norm": 13.557572364807129, "learning_rate": 8.36969696969697e-06, "num_tokens": 1017133.0, "completions/mean_length": 33.75, "completions/min_length": 30.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.75, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.35244107246398926, "rewards/meter/std": 0.4081370532512665, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9980930089950562, "rewards/repeat_soft/std": 0.0035104965791106224, "rewards/judge_quality/mean": 0.8575000166893005, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.6656577587127686, "rewards/total_composite/std": 0.19533684849739075, "reward": 0.6656577587127686, "reward_std": 0.19533684849739075, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13245141506195068, "sampling/sampling_logp_difference/max": 1.9126148223876953, "sampling/importance_sampling_ratio/min": 0.1476936787366867, "sampling/importance_sampling_ratio/mean": 0.9946917295455933, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8152019456028938, "clip_ratio/low_mean": 0.04970091744326055, "clip_ratio/low_min": 0.04970091744326055, "clip_ratio/high_mean": 0.04112903214991093, "clip_ratio/high_max": 0.04112903214991093, "clip_ratio/region_mean": 0.09082994959317148, "reward_total_mean": 0.6656577587127686, "reward_meter_mean": 0.35244107246398926, "reward_meter_std": 0.4081370532512665, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9980930089950562, "reward_repeat_soft_std": 0.0035104965791106224, "reward_judge_quality_mean": 0.8575000166893005, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.6656577587127686, "reward_total_composite_std": 0.19533684849739075} {"timestamp_utc": "2026-04-12T23:31:32Z", "mode": "train", "global_step": 540, "epoch": 0.05424409844299347, "loss": -0.0762, "grad_norm": 2.41308331489563, "learning_rate": 8.366666666666667e-06, "num_tokens": 1018520.0, "completions/mean_length": 83.375, "completions/min_length": 19.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 22.142858505249023, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.7597314119338989, "rewards/meter/std": 0.39923322200775146, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.6312500238418579, "rewards/judge_quality/std": 0.33417007327079773, "rewards/total_composite/mean": 0.7447709441184998, "rewards/total_composite/std": 0.32027193903923035, "reward": 0.7447709441184998, "reward_std": 0.32027193903923035, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.228648841381073, "sampling/sampling_logp_difference/max": 1.7817959785461426, "sampling/importance_sampling_ratio/min": 0.16833555698394775, "sampling/importance_sampling_ratio/mean": 1.0013841390609741, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5020103752613068, "clip_ratio/low_mean": 0.019736841320991516, "clip_ratio/low_min": 0.019736841320991516, "clip_ratio/high_mean": 0.15358835272490978, "clip_ratio/high_max": 0.15358835272490978, "clip_ratio/region_mean": 0.1733251940459013, "reward_total_mean": 0.7447709441184998, "reward_meter_mean": 0.7597314119338989, "reward_meter_std": 0.39923322200775146, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.6312500238418579, "reward_judge_quality_std": 0.33417007327079773, "reward_total_composite_mean": 0.7447709441184998, "reward_total_composite_std": 0.32027193903923035} {"timestamp_utc": "2026-04-12T23:31:39Z", "mode": "train", "global_step": 541, "epoch": 0.05434455047714716, "loss": -0.0138, "grad_norm": 21.918062210083008, "learning_rate": 8.363636363636365e-06, "num_tokens": 1020099.0, "completions/mean_length": 43.375, "completions/min_length": 34.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.375, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.5032355785369873, "rewards/meter/std": 0.37289944291114807, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9964313507080078, "rewards/repeat_soft/std": 0.004034888464957476, "rewards/judge_quality/mean": 0.5687500238418579, "rewards/judge_quality/std": 0.19773270189762115, "rewards/total_composite/mean": 0.6467241048812866, "rewards/total_composite/std": 0.21146126091480255, "reward": 0.6467241048812866, "reward_std": 0.21146126091480255, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17359325289726257, "sampling/sampling_logp_difference/max": 1.5767369270324707, "sampling/importance_sampling_ratio/min": 0.2066483050584793, "sampling/importance_sampling_ratio/mean": 0.9899888038635254, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9940418601036072, "clip_ratio/low_mean": 0.08224239107221365, "clip_ratio/low_min": 0.08224239107221365, "clip_ratio/high_mean": 0.09202702902257442, "clip_ratio/high_max": 0.09202702902257442, "clip_ratio/region_mean": 0.17426942009478807, "reward_total_mean": 0.6467241048812866, "reward_meter_mean": 0.5032355785369873, "reward_meter_std": 0.37289944291114807, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9964313507080078, "reward_repeat_soft_std": 0.004034888464957476, "reward_judge_quality_mean": 0.5687500238418579, "reward_judge_quality_std": 0.19773270189762115, "reward_total_composite_mean": 0.6467241048812866, "reward_total_composite_std": 0.21146126091480255} {"timestamp_utc": "2026-04-12T23:31:51Z", "mode": "train", "global_step": 542, "epoch": 0.054445002511300855, "loss": -0.0068, "grad_norm": 12.584367752075195, "learning_rate": 8.360606060606062e-06, "num_tokens": 1021703.0, "completions/mean_length": 37.5, "completions/min_length": 32.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.5, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.6922342777252197, "rewards/meter/std": 0.34142112731933594, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9844272136688232, "rewards/repeat_soft/std": 0.01921370066702366, "rewards/judge_quality/mean": 0.48124998807907104, "rewards/judge_quality/std": 0.23503421247005463, "rewards/total_composite/mean": 0.7043231725692749, "rewards/total_composite/std": 0.19310179352760315, "reward": 0.7043231725692749, "reward_std": 0.19310179352760315, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1817316710948944, "sampling/sampling_logp_difference/max": 1.6331548690795898, "sampling/importance_sampling_ratio/min": 0.19531239569187164, "sampling/importance_sampling_ratio/mean": 1.0449252128601074, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6957476437091827, "clip_ratio/low_mean": 0.05125559074804187, "clip_ratio/low_min": 0.05125559074804187, "clip_ratio/high_mean": 0.11335400026291609, "clip_ratio/high_max": 0.11335400026291609, "clip_ratio/region_mean": 0.16460959101095796, "reward_total_mean": 0.7043231725692749, "reward_meter_mean": 0.6922342777252197, "reward_meter_std": 0.34142112731933594, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9844272136688232, "reward_repeat_soft_std": 0.01921370066702366, "reward_judge_quality_mean": 0.48124998807907104, "reward_judge_quality_std": 0.23503421247005463, "reward_total_composite_mean": 0.7043231725692749, "reward_total_composite_std": 0.19310179352760315} {"timestamp_utc": "2026-04-12T23:32:03Z", "mode": "train", "global_step": 543, "epoch": 0.05454545454545454, "loss": -0.0616, "grad_norm": 3.4114131927490234, "learning_rate": 8.357575757575759e-06, "num_tokens": 1023174.0, "completions/mean_length": 89.875, "completions/min_length": 25.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 29.571430206298828, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.22027505934238434, "rewards/meter/std": 0.319637268781662, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9979627132415771, "rewards/repeat_soft/std": 0.003941006492823362, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.2569567859172821, "rewards/total_composite/mean": 0.43972378969192505, "rewards/total_composite/std": 0.23872755467891693, "reward": 0.43972378969192505, "reward_std": 0.23872753977775574, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18863677978515625, "sampling/sampling_logp_difference/max": 1.9497768878936768, "sampling/importance_sampling_ratio/min": 0.14230580627918243, "sampling/importance_sampling_ratio/mean": 1.01105535030365, "sampling/importance_sampling_ratio/max": 1.9395469427108765, "entropy": 1.2981986850500107, "clip_ratio/low_mean": 0.08865917380899191, "clip_ratio/low_min": 0.08865917380899191, "clip_ratio/high_mean": 0.10294117964804173, "clip_ratio/high_max": 0.10294117964804173, "clip_ratio/region_mean": 0.19160035345703363, "reward_total_mean": 0.43972378969192505, "reward_meter_mean": 0.22027505934238434, "reward_meter_std": 0.319637268781662, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9979627132415771, "reward_repeat_soft_std": 0.003941006492823362, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.2569567859172821, "reward_total_composite_mean": 0.43972378969192505, "reward_total_composite_std": 0.23872755467891693} {"timestamp_utc": "2026-04-12T23:32:09Z", "mode": "train", "global_step": 544, "epoch": 0.05464590657960824, "loss": -0.0043, "grad_norm": 21.975421905517578, "learning_rate": 8.354545454545455e-06, "num_tokens": 1024665.0, "completions/mean_length": 22.375, "completions/min_length": 17.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.375, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.49866294860839844, "rewards/meter/std": 0.42543089389801025, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.5936483144760132, "rewards/total_composite/std": 0.1833580583333969, "reward": 0.5936483144760132, "reward_std": 0.18335804343223572, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19714678823947906, "sampling/sampling_logp_difference/max": 0.9831533432006836, "sampling/importance_sampling_ratio/min": 0.3741294741630554, "sampling/importance_sampling_ratio/mean": 1.0339899063110352, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9947384744882584, "clip_ratio/low_mean": 0.10105444444343448, "clip_ratio/low_min": 0.10105444444343448, "clip_ratio/high_mean": 0.0641531441360712, "clip_ratio/high_max": 0.0641531441360712, "clip_ratio/region_mean": 0.16520758857950568, "reward_total_mean": 0.5936483144760132, "reward_meter_mean": 0.49866294860839844, "reward_meter_std": 0.42543089389801025, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.5936483144760132, "reward_total_composite_std": 0.1833580583333969} {"timestamp_utc": "2026-04-12T23:32:16Z", "mode": "train", "global_step": 545, "epoch": 0.05474635861376193, "loss": -0.0091, "grad_norm": 19.75861358642578, "learning_rate": 8.351515151515152e-06, "num_tokens": 1026276.0, "completions/mean_length": 37.375, "completions/min_length": 22.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.375, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.18664097785949707, "rewards/meter/std": 0.20938895642757416, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.978095531463623, "rewards/repeat_soft/std": 0.026857102289795876, "rewards/judge_quality/mean": 0.6274999976158142, "rewards/judge_quality/std": 0.21952873468399048, "rewards/total_composite/mean": 0.5200479626655579, "rewards/total_composite/std": 0.06382209807634354, "reward": 0.5200479626655579, "reward_std": 0.06382209807634354, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18388012051582336, "sampling/sampling_logp_difference/max": 1.0636630058288574, "sampling/importance_sampling_ratio/min": 0.3861510455608368, "sampling/importance_sampling_ratio/mean": 1.0321018695831299, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.381963111460209, "clip_ratio/low_mean": 0.07839714176952839, "clip_ratio/low_min": 0.07839714176952839, "clip_ratio/high_mean": 0.07873640581965446, "clip_ratio/high_max": 0.07873640581965446, "clip_ratio/region_mean": 0.15713354758918285, "reward_total_mean": 0.5200479626655579, "reward_meter_mean": 0.18664097785949707, "reward_meter_std": 0.20938895642757416, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.978095531463623, "reward_repeat_soft_std": 0.026857102289795876, "reward_judge_quality_mean": 0.6274999976158142, "reward_judge_quality_std": 0.21952873468399048, "reward_total_composite_mean": 0.5200479626655579, "reward_total_composite_std": 0.06382209807634354} {"timestamp_utc": "2026-04-12T23:32:22Z", "mode": "train", "global_step": 546, "epoch": 0.05484681064791562, "loss": 0.0109, "grad_norm": 20.110126495361328, "learning_rate": 8.348484848484849e-06, "num_tokens": 1027672.0, "completions/mean_length": 23.5, "completions/min_length": 20.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.5, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9345507621765137, "rewards/meter/std": 0.08999139815568924, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9610613584518433, "rewards/repeat_soft/std": 0.004068940877914429, "rewards/judge_quality/mean": 0.44624999165534973, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8005289435386658, "rewards/total_composite/std": 0.040175192058086395, "reward": 0.8005289435386658, "reward_std": 0.0401751846075058, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14310158789157867, "sampling/sampling_logp_difference/max": 1.5330183506011963, "sampling/importance_sampling_ratio/min": 0.2158830761909485, "sampling/importance_sampling_ratio/mean": 1.0288089513778687, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.290798395872116, "clip_ratio/low_mean": 0.045254564844071865, "clip_ratio/low_min": 0.045254564844071865, "clip_ratio/high_mean": 0.09886904759332538, "clip_ratio/high_max": 0.09886904759332538, "clip_ratio/region_mean": 0.14412361243739724, "reward_total_mean": 0.8005289435386658, "reward_meter_mean": 0.9345507621765137, "reward_meter_std": 0.08999139815568924, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9610613584518433, "reward_repeat_soft_std": 0.004068940877914429, "reward_judge_quality_mean": 0.44624999165534973, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8005289435386658, "reward_total_composite_std": 0.040175192058086395} {"timestamp_utc": "2026-04-12T23:32:28Z", "mode": "train", "global_step": 547, "epoch": 0.054947262682069314, "loss": 0.0229, "grad_norm": 20.34267234802246, "learning_rate": 8.345454545454546e-06, "num_tokens": 1029417.0, "completions/mean_length": 41.125, "completions/min_length": 32.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.125, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.8686623573303223, "rewards/meter/std": 0.33295783400535583, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.984539270401001, "rewards/repeat_soft/std": 0.01977597549557686, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7721019983291626, "rewards/total_composite/std": 0.14874032139778137, "reward": 0.7721019983291626, "reward_std": 0.14874032139778137, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1925211250782013, "sampling/sampling_logp_difference/max": 1.103989601135254, "sampling/importance_sampling_ratio/min": 0.33154571056365967, "sampling/importance_sampling_ratio/mean": 1.0144912004470825, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.628593385219574, "clip_ratio/low_mean": 0.025641025975346565, "clip_ratio/low_min": 0.025641025975346565, "clip_ratio/high_mean": 0.1589017380028963, "clip_ratio/high_max": 0.1589017380028963, "clip_ratio/region_mean": 0.18454276397824287, "reward_total_mean": 0.7721019983291626, "reward_meter_mean": 0.8686623573303223, "reward_meter_std": 0.33295783400535583, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.984539270401001, "reward_repeat_soft_std": 0.01977597549557686, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7721019983291626, "reward_total_composite_std": 0.14874032139778137} {"timestamp_utc": "2026-04-12T23:32:34Z", "mode": "train", "global_step": 548, "epoch": 0.055047714716223, "loss": 0.0271, "grad_norm": 14.278536796569824, "learning_rate": 8.342424242424244e-06, "num_tokens": 1030984.0, "completions/mean_length": 37.875, "completions/min_length": 35.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.875, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.48211273550987244, "rewards/meter/std": 0.4126070737838745, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 1.0, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.9200000166893005, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7429507374763489, "rewards/total_composite/std": 0.18567319214344025, "reward": 0.7429507374763489, "reward_std": 0.18567319214344025, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11623425781726837, "sampling/sampling_logp_difference/max": 1.375166893005371, "sampling/importance_sampling_ratio/min": 0.252797394990921, "sampling/importance_sampling_ratio/mean": 1.011572003364563, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7050644531846046, "clip_ratio/low_mean": 0.049733680207282305, "clip_ratio/low_min": 0.049733680207282305, "clip_ratio/high_mean": 0.056093559600412846, "clip_ratio/high_max": 0.056093559600412846, "clip_ratio/region_mean": 0.10582723980769515, "reward_total_mean": 0.7429507374763489, "reward_meter_mean": 0.48211273550987244, "reward_meter_std": 0.4126070737838745, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 1.0, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.9200000166893005, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7429507374763489, "reward_total_composite_std": 0.18567319214344025} {"timestamp_utc": "2026-04-12T23:32:40Z", "mode": "train", "global_step": 549, "epoch": 0.0551481667503767, "loss": 0.0343, "grad_norm": 12.85828685760498, "learning_rate": 8.339393939393941e-06, "num_tokens": 1032949.0, "completions/mean_length": 61.625, "completions/min_length": 51.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.625, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.5397204160690308, "rewards/meter/std": 0.2769630253314972, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.982119083404541, "rewards/repeat_soft/std": 0.015242212451994419, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.5807110667228699, "rewards/total_composite/std": 0.14322033524513245, "reward": 0.5807110667228699, "reward_std": 0.14322032034397125, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21319423615932465, "sampling/sampling_logp_difference/max": 1.9215459823608398, "sampling/importance_sampling_ratio/min": 0.1463804841041565, "sampling/importance_sampling_ratio/mean": 1.0339252948760986, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6116700395941734, "clip_ratio/low_mean": 0.1026963796466589, "clip_ratio/low_min": 0.1026963796466589, "clip_ratio/high_mean": 0.06415470317006111, "clip_ratio/high_max": 0.06415470317006111, "clip_ratio/region_mean": 0.16685108281672, "reward_total_mean": 0.5807110667228699, "reward_meter_mean": 0.5397204160690308, "reward_meter_std": 0.2769630253314972, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.982119083404541, "reward_repeat_soft_std": 0.015242212451994419, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.5807110667228699, "reward_total_composite_std": 0.14322033524513245} {"timestamp_utc": "2026-04-12T23:32:52Z", "mode": "train", "global_step": 550, "epoch": 0.055248618784530384, "loss": -0.132, "grad_norm": 2.678556442260742, "learning_rate": 8.336363636363636e-06, "num_tokens": 1034657.0, "completions/mean_length": 108.5, "completions/min_length": 39.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 50.85714340209961, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9733370542526245, "rewards/meter/std": 0.03384688124060631, "rewards/count_adherence/mean": 0.7916666865348816, "rewards/count_adherence/std": 0.17251639068126678, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9945548176765442, "rewards/repeat_soft/std": 0.005410619545727968, "rewards/judge_quality/mean": 0.5800000429153442, "rewards/judge_quality/std": 0.25595760345458984, "rewards/total_composite/mean": 0.830207109451294, "rewards/total_composite/std": 0.0916534811258316, "reward": 0.830207109451294, "reward_std": 0.09165346622467041, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18270613253116608, "sampling/sampling_logp_difference/max": 1.4461555480957031, "sampling/importance_sampling_ratio/min": 0.23547382652759552, "sampling/importance_sampling_ratio/mean": 1.037292718887329, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5511703938245773, "clip_ratio/low_mean": 0.008064515888690948, "clip_ratio/low_min": 0.008064515888690948, "clip_ratio/high_mean": 0.11189078446477652, "clip_ratio/high_max": 0.11189078446477652, "clip_ratio/region_mean": 0.11995530035346746, "reward_total_mean": 0.830207109451294, "reward_meter_mean": 0.9733370542526245, "reward_meter_std": 0.03384688124060631, "reward_count_adherence_mean": 0.7916666865348816, "reward_count_adherence_std": 0.17251639068126678, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9945548176765442, "reward_repeat_soft_std": 0.005410619545727968, "reward_judge_quality_mean": 0.5800000429153442, "reward_judge_quality_std": 0.25595760345458984, "reward_total_composite_mean": 0.830207109451294, "reward_total_composite_std": 0.0916534811258316} {"timestamp_utc": "2026-04-12T23:33:30Z", "mode": "eval", "global_step": 550, "epoch": 0.055248618784530384, "eval_loss": NaN, "eval_runtime": 38.0955, "eval_samples_per_second": 2.1, "eval_steps_per_second": 0.262, "eval_num_tokens": 1034657.0, "eval_completions/mean_length": 52.95, "eval_completions/min_length": 28.5, "eval_completions/max_length": 90.8, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 52.95, "eval_completions/min_terminated_length": 28.5, "eval_completions/max_terminated_length": 90.8, "eval_rewards/meter/mean": 0.6768401026725769, "eval_rewards/meter/std": 0.3547126233577728, "eval_rewards/count_adherence/mean": 0.8618750035762787, "eval_rewards/count_adherence/std": 0.1260939285159111, "eval_rewards/hard_gate/mean": 0.9875, "eval_rewards/hard_gate/std": 0.03535533845424652, "eval_rewards/repeat_soft/mean": 0.9897583186626434, "eval_rewards/repeat_soft/std": 0.0133729699999094, "eval_rewards/judge_quality/mean": 0.5565000027418137, "eval_rewards/judge_quality/std": 0.21696364432573317, "eval_rewards/total_composite/mean": 0.6896012306213379, "eval_rewards/total_composite/std": 0.1894986227154732, "eval_reward": 0.6896012306213379, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.11231597661972045, "eval_sampling/sampling_logp_difference/max": 1.0312015533447265, "eval_sampling/importance_sampling_ratio/min": 0.3648092597723007, "eval_sampling/importance_sampling_ratio/mean": 1.0298011302947998, "eval_sampling/importance_sampling_ratio/max": 1.514595627784729, "eval_entropy": 1.5228630781173706, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6896012306213379, "eval_reward_meter_mean": 0.6768401026725769, "eval_reward_meter_std": 0.3547126233577728, "eval_reward_count_adherence_mean": 0.8618750035762787, "eval_reward_count_adherence_std": 0.1260939285159111, "eval_reward_hard_gate_mean": 0.9875, "eval_reward_hard_gate_std": 0.03535533845424652, "eval_reward_repeat_soft_mean": 0.9897583186626434, "eval_reward_repeat_soft_std": 0.0133729699999094, "eval_reward_judge_quality_mean": 0.5565000027418137, "eval_reward_judge_quality_std": 0.21696364432573317, "eval_reward_total_composite_mean": 0.6896012306213379, "eval_reward_total_composite_std": 0.1894986227154732} {"timestamp_utc": "2026-04-12T23:33:39Z", "mode": "train", "global_step": 551, "epoch": 0.05534907081868408, "loss": 0.0654, "grad_norm": 22.09537696838379, "learning_rate": 8.333333333333334e-06, "num_tokens": 1036244.0, "completions/mean_length": 36.375, "completions/min_length": 31.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.375, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.8903855085372925, "rewards/meter/std": 0.12534716725349426, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9896296262741089, "rewards/repeat_soft/std": 0.020700186491012573, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7823864221572876, "rewards/total_composite/std": 0.05416727811098099, "reward": 0.7823864221572876, "reward_std": 0.054167263209819794, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16783346235752106, "sampling/sampling_logp_difference/max": 2.0916714668273926, "sampling/importance_sampling_ratio/min": 0.12348057329654694, "sampling/importance_sampling_ratio/mean": 1.033523440361023, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2396157458424568, "clip_ratio/low_mean": 0.03709847875870764, "clip_ratio/low_min": 0.03709847875870764, "clip_ratio/high_mean": 0.08256973512470722, "clip_ratio/high_max": 0.08256973512470722, "clip_ratio/region_mean": 0.11966821388341486, "reward_total_mean": 0.7823864221572876, "reward_meter_mean": 0.8903855085372925, "reward_meter_std": 0.12534716725349426, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9896296262741089, "reward_repeat_soft_std": 0.020700186491012573, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7823864221572876, "reward_total_composite_std": 0.05416727811098099} {"timestamp_utc": "2026-04-12T23:33:47Z", "mode": "train", "global_step": 552, "epoch": 0.05544952285283777, "loss": 0.0414, "grad_norm": 16.699420928955078, "learning_rate": 8.330303030303031e-06, "num_tokens": 1037996.0, "completions/mean_length": 39.0, "completions/min_length": 33.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.8217688202857971, "rewards/meter/std": 0.2882085144519806, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9789672493934631, "rewards/repeat_soft/std": 0.023606404662132263, "rewards/judge_quality/mean": 0.6225000023841858, "rewards/judge_quality/std": 0.24656209349632263, "rewards/total_composite/mean": 0.8044426441192627, "rewards/total_composite/std": 0.16560649871826172, "reward": 0.8044426441192627, "reward_std": 0.16560646891593933, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16607393324375153, "sampling/sampling_logp_difference/max": 1.3295087814331055, "sampling/importance_sampling_ratio/min": 0.2646072208881378, "sampling/importance_sampling_ratio/mean": 1.0421935319900513, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4020631834864616, "clip_ratio/low_mean": 0.04494028817862272, "clip_ratio/low_min": 0.04494028817862272, "clip_ratio/high_mean": 0.08863737061619759, "clip_ratio/high_max": 0.08863737061619759, "clip_ratio/region_mean": 0.1335776587948203, "reward_total_mean": 0.8044426441192627, "reward_meter_mean": 0.8217688202857971, "reward_meter_std": 0.2882085144519806, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9789672493934631, "reward_repeat_soft_std": 0.023606404662132263, "reward_judge_quality_mean": 0.6225000023841858, "reward_judge_quality_std": 0.24656209349632263, "reward_total_composite_mean": 0.8044426441192627, "reward_total_composite_std": 0.16560649871826172} {"timestamp_utc": "2026-04-12T23:33:53Z", "mode": "train", "global_step": 553, "epoch": 0.05554997488699146, "loss": 0.0216, "grad_norm": 20.514137268066406, "learning_rate": 8.327272727272728e-06, "num_tokens": 1039509.0, "completions/mean_length": 30.125, "completions/min_length": 28.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.125, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.30361396074295044, "rewards/meter/std": 0.31711187958717346, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9963240623474121, "rewards/repeat_soft/std": 0.01039720606058836, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.5685086250305176, "rewards/total_composite/std": 0.14833001792430878, "reward": 0.5685086250305176, "reward_std": 0.14833003282546997, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1256062090396881, "sampling/sampling_logp_difference/max": 1.7478203773498535, "sampling/importance_sampling_ratio/min": 0.17415310442447662, "sampling/importance_sampling_ratio/mean": 1.001899242401123, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.83380376547575, "clip_ratio/low_mean": 0.05244030198082328, "clip_ratio/low_min": 0.05244030198082328, "clip_ratio/high_mean": 0.05044957436621189, "clip_ratio/high_max": 0.05044957436621189, "clip_ratio/region_mean": 0.10288987634703517, "reward_total_mean": 0.5685086250305176, "reward_meter_mean": 0.30361396074295044, "reward_meter_std": 0.31711187958717346, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9963240623474121, "reward_repeat_soft_std": 0.01039720606058836, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.5685086250305176, "reward_total_composite_std": 0.14833001792430878} {"timestamp_utc": "2026-04-12T23:33:59Z", "mode": "train", "global_step": 554, "epoch": 0.055650426921145156, "loss": -0.1198, "grad_norm": 15.291116714477539, "learning_rate": 8.324242424242425e-06, "num_tokens": 1041157.0, "completions/mean_length": 37.0, "completions/min_length": 21.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.0, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9781914949417114, "rewards/meter/std": 0.020928315818309784, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9927212595939636, "rewards/repeat_soft/std": 0.013078841380774975, "rewards/judge_quality/mean": 0.3974999785423279, "rewards/judge_quality/std": 0.2272663712501526, "rewards/total_composite/mean": 0.7899582982063293, "rewards/total_composite/std": 0.09415343403816223, "reward": 0.7899582982063293, "reward_std": 0.09415342658758163, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20310796797275543, "sampling/sampling_logp_difference/max": 2.109273672103882, "sampling/importance_sampling_ratio/min": 0.23027895390987396, "sampling/importance_sampling_ratio/mean": 1.0313785076141357, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.004178002476692, "clip_ratio/low_mean": 0.10269447229802608, "clip_ratio/low_min": 0.10269447229802608, "clip_ratio/high_mean": 0.1045262087136507, "clip_ratio/high_max": 0.1045262087136507, "clip_ratio/region_mean": 0.2072206810116768, "reward_total_mean": 0.7899582982063293, "reward_meter_mean": 0.9781914949417114, "reward_meter_std": 0.020928315818309784, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9927212595939636, "reward_repeat_soft_std": 0.013078841380774975, "reward_judge_quality_mean": 0.3974999785423279, "reward_judge_quality_std": 0.2272663712501526, "reward_total_composite_mean": 0.7899582982063293, "reward_total_composite_std": 0.09415343403816223} {"timestamp_utc": "2026-04-12T23:34:06Z", "mode": "train", "global_step": 555, "epoch": 0.055750878955298844, "loss": 0.0889, "grad_norm": 16.005897521972656, "learning_rate": 8.321212121212123e-06, "num_tokens": 1042827.0, "completions/mean_length": 35.75, "completions/min_length": 25.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.75, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.7009669542312622, "rewards/meter/std": 0.4076966345310211, "rewards/count_adherence/mean": 0.7916666865348816, "rewards/count_adherence/std": 0.17251639068126678, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9876705408096313, "rewards/repeat_soft/std": 0.014502830803394318, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.7714521884918213, "rewards/total_composite/std": 0.14572963118553162, "reward": 0.7714521884918213, "reward_std": 0.14572963118553162, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1425628662109375, "sampling/sampling_logp_difference/max": 1.3315086364746094, "sampling/importance_sampling_ratio/min": 0.2640785574913025, "sampling/importance_sampling_ratio/mean": 1.017378807067871, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2101535201072693, "clip_ratio/low_mean": 0.0983163183555007, "clip_ratio/low_min": 0.0983163183555007, "clip_ratio/high_mean": 0.06729654874652624, "clip_ratio/high_max": 0.06729654874652624, "clip_ratio/region_mean": 0.16561286710202694, "reward_total_mean": 0.7714521884918213, "reward_meter_mean": 0.7009669542312622, "reward_meter_std": 0.4076966345310211, "reward_count_adherence_mean": 0.7916666865348816, "reward_count_adherence_std": 0.17251639068126678, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9876705408096313, "reward_repeat_soft_std": 0.014502830803394318, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.7714521884918213, "reward_total_composite_std": 0.14572963118553162} {"timestamp_utc": "2026-04-12T23:34:18Z", "mode": "train", "global_step": 556, "epoch": 0.05585133098945254, "loss": 0.0668, "grad_norm": 17.48725700378418, "learning_rate": 8.318181818181818e-06, "num_tokens": 1044307.0, "completions/mean_length": 32.0, "completions/min_length": 27.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.5814489126205444, "rewards/meter/std": 0.28750914335250854, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9948940277099609, "rewards/repeat_soft/std": 0.006360895931720734, "rewards/judge_quality/mean": 0.6762499809265137, "rewards/judge_quality/std": 0.2081165760755539, "rewards/total_composite/mean": 0.7140164375305176, "rewards/total_composite/std": 0.11591407656669617, "reward": 0.7140164375305176, "reward_std": 0.11591405421495438, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15279623866081238, "sampling/sampling_logp_difference/max": 1.2136955261230469, "sampling/importance_sampling_ratio/min": 0.2970973253250122, "sampling/importance_sampling_ratio/mean": 1.0268559455871582, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.299508735537529, "clip_ratio/low_mean": 0.06589076202362776, "clip_ratio/low_min": 0.06589076202362776, "clip_ratio/high_mean": 0.05502621131017804, "clip_ratio/high_max": 0.05502621131017804, "clip_ratio/region_mean": 0.1209169733338058, "reward_total_mean": 0.7140164375305176, "reward_meter_mean": 0.5814489126205444, "reward_meter_std": 0.28750914335250854, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9948940277099609, "reward_repeat_soft_std": 0.006360895931720734, "reward_judge_quality_mean": 0.6762499809265137, "reward_judge_quality_std": 0.2081165760755539, "reward_total_composite_mean": 0.7140164375305176, "reward_total_composite_std": 0.11591407656669617} {"timestamp_utc": "2026-04-12T23:34:24Z", "mode": "train", "global_step": 557, "epoch": 0.055951783023606226, "loss": 0.0236, "grad_norm": 22.84958267211914, "learning_rate": 8.315151515151516e-06, "num_tokens": 1045817.0, "completions/mean_length": 27.75, "completions/min_length": 24.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.75, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.6492612361907959, "rewards/meter/std": 0.31114980578422546, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9975765943527222, "rewards/repeat_soft/std": 0.003681178204715252, "rewards/judge_quality/mean": 0.6187499761581421, "rewards/judge_quality/std": 0.24976776540279388, "rewards/total_composite/mean": 0.727550208568573, "rewards/total_composite/std": 0.1336294710636139, "reward": 0.727550208568573, "reward_std": 0.1336294710636139, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21018679440021515, "sampling/sampling_logp_difference/max": 2.267002820968628, "sampling/importance_sampling_ratio/min": 0.10362229496240616, "sampling/importance_sampling_ratio/mean": 1.0085163116455078, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9445214346051216, "clip_ratio/low_mean": 0.10019556060433388, "clip_ratio/low_min": 0.10019556060433388, "clip_ratio/high_mean": 0.11317965760827065, "clip_ratio/high_max": 0.11317965760827065, "clip_ratio/region_mean": 0.21337521821260452, "reward_total_mean": 0.727550208568573, "reward_meter_mean": 0.6492612361907959, "reward_meter_std": 0.31114980578422546, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9975765943527222, "reward_repeat_soft_std": 0.003681178204715252, "reward_judge_quality_mean": 0.6187499761581421, "reward_judge_quality_std": 0.24976776540279388, "reward_total_composite_mean": 0.727550208568573, "reward_total_composite_std": 0.1336294710636139} {"timestamp_utc": "2026-04-12T23:34:36Z", "mode": "train", "global_step": 558, "epoch": 0.05605223505775992, "loss": -0.0756, "grad_norm": 4.199847221374512, "learning_rate": 8.312121212121213e-06, "num_tokens": 1047211.0, "completions/mean_length": 87.25, "completions/min_length": 20.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 26.571430206298828, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.36835283041000366, "rewards/meter/std": 0.49167048931121826, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9469515085220337, "rewards/repeat_soft/std": 0.05719867721199989, "rewards/judge_quality/mean": 0.39625000953674316, "rewards/judge_quality/std": 0.14029940962791443, "rewards/total_composite/mean": 0.4962038993835449, "rewards/total_composite/std": 0.28817540407180786, "reward": 0.4962038993835449, "reward_std": 0.2881753742694855, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21869458258152008, "sampling/sampling_logp_difference/max": 1.4984703063964844, "sampling/importance_sampling_ratio/min": 0.2234717607498169, "sampling/importance_sampling_ratio/mean": 1.0446205139160156, "sampling/importance_sampling_ratio/max": 1.9030182361602783, "entropy": 1.4976290464401245, "clip_ratio/low_mean": 0.058333334513008595, "clip_ratio/low_min": 0.058333334513008595, "clip_ratio/high_mean": 0.07395362202078104, "clip_ratio/high_max": 0.07395362202078104, "clip_ratio/region_mean": 0.13228695653378963, "reward_total_mean": 0.4962038993835449, "reward_meter_mean": 0.36835283041000366, "reward_meter_std": 0.49167048931121826, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9469515085220337, "reward_repeat_soft_std": 0.05719867721199989, "reward_judge_quality_mean": 0.39625000953674316, "reward_judge_quality_std": 0.14029940962791443, "reward_total_composite_mean": 0.4962038993835449, "reward_total_composite_std": 0.28817540407180786} {"timestamp_utc": "2026-04-12T23:34:43Z", "mode": "train", "global_step": 559, "epoch": 0.05615268709191361, "loss": -0.0437, "grad_norm": 17.94271469116211, "learning_rate": 8.30909090909091e-06, "num_tokens": 1048718.0, "completions/mean_length": 31.375, "completions/min_length": 23.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.375, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.8698766827583313, "rewards/meter/std": 0.1697571724653244, "rewards/count_adherence/mean": 0.7916666865348816, "rewards/count_adherence/std": 0.17251639068126678, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9940952062606812, "rewards/repeat_soft/std": 0.007303288672119379, "rewards/judge_quality/mean": 0.6600000262260437, "rewards/judge_quality/std": 0.20099753141403198, "rewards/total_composite/mean": 0.8076040148735046, "rewards/total_composite/std": 0.09475772082805634, "reward": 0.8076040148735046, "reward_std": 0.09475770592689514, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1510203778743744, "sampling/sampling_logp_difference/max": 1.8915691375732422, "sampling/importance_sampling_ratio/min": 0.15083494782447815, "sampling/importance_sampling_ratio/mean": 0.9716265797615051, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.776003822684288, "clip_ratio/low_mean": 0.05181878339499235, "clip_ratio/low_min": 0.05181878339499235, "clip_ratio/high_mean": 0.09417243767529726, "clip_ratio/high_max": 0.09417243767529726, "clip_ratio/region_mean": 0.1459912210702896, "reward_total_mean": 0.8076040148735046, "reward_meter_mean": 0.8698766827583313, "reward_meter_std": 0.1697571724653244, "reward_count_adherence_mean": 0.7916666865348816, "reward_count_adherence_std": 0.17251639068126678, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9940952062606812, "reward_repeat_soft_std": 0.007303288672119379, "reward_judge_quality_mean": 0.6600000262260437, "reward_judge_quality_std": 0.20099753141403198, "reward_total_composite_mean": 0.8076040148735046, "reward_total_composite_std": 0.09475772082805634} {"timestamp_utc": "2026-04-12T23:34:55Z", "mode": "train", "global_step": 560, "epoch": 0.0562531391260673, "loss": -0.0505, "grad_norm": 5.914506912231445, "learning_rate": 8.306060606060606e-06, "num_tokens": 1050069.0, "completions/mean_length": 83.875, "completions/min_length": 19.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 22.71428680419922, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.8616518974304199, "rewards/meter/std": 0.3274402618408203, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9411051273345947, "rewards/repeat_soft/std": 0.027129443362355232, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.13617216050624847, "rewards/total_composite/mean": 0.6638503074645996, "rewards/total_composite/std": 0.3063156306743622, "reward": 0.6638503074645996, "reward_std": 0.3063156306743622, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1612595170736313, "sampling/sampling_logp_difference/max": 0.8563566207885742, "sampling/importance_sampling_ratio/min": 0.4247066378593445, "sampling/importance_sampling_ratio/mean": 1.0361205339431763, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2113301903009415, "clip_ratio/low_mean": 0.03448275849223137, "clip_ratio/low_min": 0.03448275849223137, "clip_ratio/high_mean": 0.11591559275984764, "clip_ratio/high_max": 0.11591559275984764, "clip_ratio/region_mean": 0.150398351252079, "reward_total_mean": 0.6638503074645996, "reward_meter_mean": 0.8616518974304199, "reward_meter_std": 0.3274402618408203, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9411051273345947, "reward_repeat_soft_std": 0.027129443362355232, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.13617216050624847, "reward_total_composite_mean": 0.6638503074645996, "reward_total_composite_std": 0.3063156306743622} {"timestamp_utc": "2026-04-12T23:35:03Z", "mode": "train", "global_step": 561, "epoch": 0.056353591160221, "loss": -0.0149, "grad_norm": 8.69936752319336, "learning_rate": 8.303030303030305e-06, "num_tokens": 1052575.0, "completions/mean_length": 105.25, "completions/min_length": 83.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.25, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.8106032013893127, "rewards/meter/std": 0.3232221305370331, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.982526421546936, "rewards/repeat_soft/std": 0.020498819649219513, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.21357084810733795, "rewards/total_composite/mean": 0.7363990545272827, "rewards/total_composite/std": 0.18422317504882812, "reward": 0.7363990545272827, "reward_std": 0.18422316014766693, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20344385504722595, "sampling/sampling_logp_difference/max": 1.874476432800293, "sampling/importance_sampling_ratio/min": 0.15343527495861053, "sampling/importance_sampling_ratio/mean": 1.029186487197876, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.249633803963661, "clip_ratio/low_mean": 0.04466950334608555, "clip_ratio/low_min": 0.04466950334608555, "clip_ratio/high_mean": 0.15353423915803432, "clip_ratio/high_max": 0.15353423915803432, "clip_ratio/region_mean": 0.19820374250411987, "reward_total_mean": 0.7363990545272827, "reward_meter_mean": 0.8106032013893127, "reward_meter_std": 0.3232221305370331, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.982526421546936, "reward_repeat_soft_std": 0.020498819649219513, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.21357084810733795, "reward_total_composite_mean": 0.7363990545272827, "reward_total_composite_std": 0.18422317504882812} {"timestamp_utc": "2026-04-12T23:35:10Z", "mode": "train", "global_step": 562, "epoch": 0.056454043194374685, "loss": 0.0406, "grad_norm": 17.79619789123535, "learning_rate": 8.3e-06, "num_tokens": 1054121.0, "completions/mean_length": 36.25, "completions/min_length": 32.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.25, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.4195706844329834, "rewards/meter/std": 0.31495755910873413, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9929752349853516, "rewards/repeat_soft/std": 0.007955558598041534, "rewards/judge_quality/mean": 0.6850000023841858, "rewards/judge_quality/std": 0.25961512327194214, "rewards/total_composite/mean": 0.6436043381690979, "rewards/total_composite/std": 0.18055899441242218, "reward": 0.6436043381690979, "reward_std": 0.18055900931358337, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14549636840820312, "sampling/sampling_logp_difference/max": 1.401745319366455, "sampling/importance_sampling_ratio/min": 0.24616694450378418, "sampling/importance_sampling_ratio/mean": 1.0097100734710693, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.04088294506073, "clip_ratio/low_mean": 0.06676417216658592, "clip_ratio/low_min": 0.06676417216658592, "clip_ratio/high_mean": 0.06196917872875929, "clip_ratio/high_max": 0.06196917872875929, "clip_ratio/region_mean": 0.1287333508953452, "reward_total_mean": 0.6436043381690979, "reward_meter_mean": 0.4195706844329834, "reward_meter_std": 0.31495755910873413, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9929752349853516, "reward_repeat_soft_std": 0.007955558598041534, "reward_judge_quality_mean": 0.6850000023841858, "reward_judge_quality_std": 0.25961512327194214, "reward_total_composite_mean": 0.6436043381690979, "reward_total_composite_std": 0.18055899441242218} {"timestamp_utc": "2026-04-12T23:35:18Z", "mode": "train", "global_step": 563, "epoch": 0.05655449522852838, "loss": 0.0074, "grad_norm": 14.349466323852539, "learning_rate": 8.296969696969697e-06, "num_tokens": 1055876.0, "completions/mean_length": 45.375, "completions/min_length": 41.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.375, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.28119686245918274, "rewards/meter/std": 0.2920469045639038, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.993741512298584, "rewards/repeat_soft/std": 0.008978321217000484, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.45528775453567505, "rewards/total_composite/std": 0.1296423375606537, "reward": 0.45528775453567505, "reward_std": 0.12964235246181488, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14609691500663757, "sampling/sampling_logp_difference/max": 2.6330089569091797, "sampling/importance_sampling_ratio/min": 0.07186190783977509, "sampling/importance_sampling_ratio/mean": 1.0085500478744507, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7967784255743027, "clip_ratio/low_mean": 0.09545139782130718, "clip_ratio/low_min": 0.09545139782130718, "clip_ratio/high_mean": 0.031816122122108936, "clip_ratio/high_max": 0.031816122122108936, "clip_ratio/region_mean": 0.12726751994341612, "reward_total_mean": 0.45528775453567505, "reward_meter_mean": 0.28119686245918274, "reward_meter_std": 0.2920469045639038, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.993741512298584, "reward_repeat_soft_std": 0.008978321217000484, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.45528775453567505, "reward_total_composite_std": 0.1296423375606537} {"timestamp_utc": "2026-04-12T23:35:25Z", "mode": "train", "global_step": 564, "epoch": 0.05665494726268207, "loss": 0.1616, "grad_norm": 25.077064514160156, "learning_rate": 8.293939393939395e-06, "num_tokens": 1057384.0, "completions/mean_length": 22.5, "completions/min_length": 15.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.5, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.3891432583332062, "rewards/meter/std": 0.4450378119945526, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9594065546989441, "rewards/repeat_soft/std": 0.008749538101255894, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5504301190376282, "rewards/total_composite/std": 0.1980075240135193, "reward": 0.5504301190376282, "reward_std": 0.19800753891468048, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15745225548744202, "sampling/sampling_logp_difference/max": 1.043484091758728, "sampling/importance_sampling_ratio/min": 0.3522253632545471, "sampling/importance_sampling_ratio/mean": 1.0057356357574463, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2190730944275856, "clip_ratio/low_mean": 0.08448616694658995, "clip_ratio/low_min": 0.08448616694658995, "clip_ratio/high_mean": 0.05454545561224222, "clip_ratio/high_max": 0.05454545561224222, "clip_ratio/region_mean": 0.13903162255883217, "reward_total_mean": 0.5504301190376282, "reward_meter_mean": 0.3891432583332062, "reward_meter_std": 0.4450378119945526, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9594065546989441, "reward_repeat_soft_std": 0.008749538101255894, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5504301190376282, "reward_total_composite_std": 0.1980075240135193} {"timestamp_utc": "2026-04-12T23:35:32Z", "mode": "train", "global_step": 565, "epoch": 0.05675539929683576, "loss": 0.0632, "grad_norm": 25.794031143188477, "learning_rate": 8.290909090909092e-06, "num_tokens": 1058806.0, "completions/mean_length": 22.75, "completions/min_length": 20.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.75, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.6978994607925415, "rewards/meter/std": 0.4174855053424835, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9258180856704712, "rewards/repeat_soft/std": 0.06930955499410629, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.6837615370750427, "rewards/total_composite/std": 0.1859247088432312, "reward": 0.6837615370750427, "reward_std": 0.1859247088432312, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15580487251281738, "sampling/sampling_logp_difference/max": 1.7292284965515137, "sampling/importance_sampling_ratio/min": 0.17742124199867249, "sampling/importance_sampling_ratio/mean": 1.0350288152694702, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2899454236030579, "clip_ratio/low_mean": 0.04881599359214306, "clip_ratio/low_min": 0.04881599359214306, "clip_ratio/high_mean": 0.07444279920309782, "clip_ratio/high_max": 0.07444279920309782, "clip_ratio/region_mean": 0.12325879279524088, "reward_total_mean": 0.6837615370750427, "reward_meter_mean": 0.6978994607925415, "reward_meter_std": 0.4174855053424835, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9258180856704712, "reward_repeat_soft_std": 0.06930955499410629, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.6837615370750427, "reward_total_composite_std": 0.1859247088432312} {"timestamp_utc": "2026-04-12T23:35:38Z", "mode": "train", "global_step": 566, "epoch": 0.05685585133098945, "loss": 0.0616, "grad_norm": 14.408302307128906, "learning_rate": 8.287878787878787e-06, "num_tokens": 1060294.0, "completions/mean_length": 42.0, "completions/min_length": 38.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.0, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.6537537574768066, "rewards/meter/std": 0.45629772543907166, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9942512512207031, "rewards/repeat_soft/std": 0.009806578978896141, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.6808643341064453, "rewards/total_composite/std": 0.21611088514328003, "reward": 0.6808643341064453, "reward_std": 0.21611087024211884, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18364323675632477, "sampling/sampling_logp_difference/max": 1.5394468307495117, "sampling/importance_sampling_ratio/min": 0.21449974179267883, "sampling/importance_sampling_ratio/mean": 1.0270146131515503, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7453376650810242, "clip_ratio/low_mean": 0.06401203386485577, "clip_ratio/low_min": 0.06401203386485577, "clip_ratio/high_mean": 0.13138397689908743, "clip_ratio/high_max": 0.13138397689908743, "clip_ratio/region_mean": 0.1953960107639432, "reward_total_mean": 0.6808643341064453, "reward_meter_mean": 0.6537537574768066, "reward_meter_std": 0.45629772543907166, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9942512512207031, "reward_repeat_soft_std": 0.009806578978896141, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.6808643341064453, "reward_total_composite_std": 0.21611088514328003} {"timestamp_utc": "2026-04-12T23:35:46Z", "mode": "train", "global_step": 567, "epoch": 0.056956303365143145, "loss": 0.0032, "grad_norm": 16.321535110473633, "learning_rate": 8.284848484848486e-06, "num_tokens": 1061830.0, "completions/mean_length": 41.0, "completions/min_length": 34.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.8697830438613892, "rewards/meter/std": 0.27155664563179016, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.998328447341919, "rewards/repeat_soft/std": 0.0024800391402095556, "rewards/judge_quality/mean": 0.7224999666213989, "rewards/judge_quality/std": 0.2510406970977783, "rewards/total_composite/mean": 0.8579851984977722, "rewards/total_composite/std": 0.1681470274925232, "reward": 0.8579851984977722, "reward_std": 0.1681470423936844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15238729119300842, "sampling/sampling_logp_difference/max": 1.2545852661132812, "sampling/importance_sampling_ratio/min": 0.2851940989494324, "sampling/importance_sampling_ratio/mean": 1.0231162309646606, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.245242677628994, "clip_ratio/low_mean": 0.06068353448063135, "clip_ratio/low_min": 0.06068353448063135, "clip_ratio/high_mean": 0.07750655431300402, "clip_ratio/high_max": 0.07750655431300402, "clip_ratio/region_mean": 0.13819008879363537, "reward_total_mean": 0.8579851984977722, "reward_meter_mean": 0.8697830438613892, "reward_meter_std": 0.27155664563179016, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.998328447341919, "reward_repeat_soft_std": 0.0024800391402095556, "reward_judge_quality_mean": 0.7224999666213989, "reward_judge_quality_std": 0.2510406970977783, "reward_total_composite_mean": 0.8579851984977722, "reward_total_composite_std": 0.1681470274925232} {"timestamp_utc": "2026-04-12T23:35:55Z", "mode": "train", "global_step": 568, "epoch": 0.05705675539929683, "loss": 0.0991, "grad_norm": 23.553218841552734, "learning_rate": 8.281818181818182e-06, "num_tokens": 1063597.0, "completions/mean_length": 41.875, "completions/min_length": 36.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.875, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.38990768790245056, "rewards/meter/std": 0.4131592810153961, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9963082671165466, "rewards/repeat_soft/std": 0.005969129037111998, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.5198392868041992, "rewards/total_composite/std": 0.17948034405708313, "reward": 0.5198392868041992, "reward_std": 0.17948034405708313, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17765597999095917, "sampling/sampling_logp_difference/max": 1.7947359085083008, "sampling/importance_sampling_ratio/min": 0.16617132723331451, "sampling/importance_sampling_ratio/mean": 1.0089542865753174, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2370899990200996, "clip_ratio/low_mean": 0.07959414646029472, "clip_ratio/low_min": 0.07959414646029472, "clip_ratio/high_mean": 0.08506188355386257, "clip_ratio/high_max": 0.08506188355386257, "clip_ratio/region_mean": 0.1646560300141573, "reward_total_mean": 0.5198392868041992, "reward_meter_mean": 0.38990768790245056, "reward_meter_std": 0.4131592810153961, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9963082671165466, "reward_repeat_soft_std": 0.005969129037111998, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.5198392868041992, "reward_total_composite_std": 0.17948034405708313} {"timestamp_utc": "2026-04-12T23:36:00Z", "mode": "train", "global_step": 569, "epoch": 0.05715720743345053, "loss": 0.0082, "grad_norm": 25.268598556518555, "learning_rate": 8.27878787878788e-06, "num_tokens": 1065005.0, "completions/mean_length": 26.0, "completions/min_length": 20.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.0, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.7547222375869751, "rewards/meter/std": 0.29855018854141235, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9790502786636353, "rewards/repeat_soft/std": 0.006777657195925713, "rewards/judge_quality/mean": 0.8725000023841858, "rewards/judge_quality/std": 0.18343937397003174, "rewards/total_composite/mean": 0.8492799997329712, "rewards/total_composite/std": 0.1814108043909073, "reward": 0.8492799997329712, "reward_std": 0.1814108043909073, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09628446400165558, "sampling/sampling_logp_difference/max": 1.2104530334472656, "sampling/importance_sampling_ratio/min": 0.29806220531463623, "sampling/importance_sampling_ratio/mean": 0.982908308506012, "sampling/importance_sampling_ratio/max": 1.7888442277908325, "entropy": 0.5408477783203125, "clip_ratio/low_mean": 0.02446428546682, "clip_ratio/low_min": 0.02446428546682, "clip_ratio/high_mean": 0.07273707026615739, "clip_ratio/high_max": 0.07273707026615739, "clip_ratio/region_mean": 0.09720135573297739, "reward_total_mean": 0.8492799997329712, "reward_meter_mean": 0.7547222375869751, "reward_meter_std": 0.29855018854141235, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9790502786636353, "reward_repeat_soft_std": 0.006777657195925713, "reward_judge_quality_mean": 0.8725000023841858, "reward_judge_quality_std": 0.18343937397003174, "reward_total_composite_mean": 0.8492799997329712, "reward_total_composite_std": 0.1814108043909073} {"timestamp_utc": "2026-04-12T23:36:07Z", "mode": "train", "global_step": 570, "epoch": 0.05725765946760422, "loss": 0.0576, "grad_norm": 17.673782348632812, "learning_rate": 8.275757575757577e-06, "num_tokens": 1066588.0, "completions/mean_length": 40.875, "completions/min_length": 34.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.875, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.5464320182800293, "rewards/meter/std": 0.3933197855949402, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9748578667640686, "rewards/repeat_soft/std": 0.040405649691820145, "rewards/judge_quality/mean": 0.45499998331069946, "rewards/judge_quality/std": 0.12739142775535583, "rewards/total_composite/mean": 0.5873031616210938, "rewards/total_composite/std": 0.27691811323165894, "reward": 0.5873031616210938, "reward_std": 0.27691811323165894, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18612347543239594, "sampling/sampling_logp_difference/max": 1.3434853553771973, "sampling/importance_sampling_ratio/min": 0.26093465089797974, "sampling/importance_sampling_ratio/mean": 1.0281500816345215, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7049905508756638, "clip_ratio/low_mean": 0.04561272170394659, "clip_ratio/low_min": 0.04561272170394659, "clip_ratio/high_mean": 0.1491043008863926, "clip_ratio/high_max": 0.1491043008863926, "clip_ratio/region_mean": 0.19471702259033918, "reward_total_mean": 0.5873031616210938, "reward_meter_mean": 0.5464320182800293, "reward_meter_std": 0.3933197855949402, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9748578667640686, "reward_repeat_soft_std": 0.040405649691820145, "reward_judge_quality_mean": 0.45499998331069946, "reward_judge_quality_std": 0.12739142775535583, "reward_total_composite_mean": 0.5873031616210938, "reward_total_composite_std": 0.27691811323165894} {"timestamp_utc": "2026-04-12T23:36:13Z", "mode": "train", "global_step": 571, "epoch": 0.05735811150175791, "loss": -0.0195, "grad_norm": 9.72391128540039, "learning_rate": 8.272727272727274e-06, "num_tokens": 1069016.0, "completions/mean_length": 90.5, "completions/min_length": 78.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.5, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.7412514686584473, "rewards/meter/std": 0.28295058012008667, "rewards/count_adherence/mean": 0.7749999761581421, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9967364072799683, "rewards/repeat_soft/std": 0.0038352853152900934, "rewards/judge_quality/mean": 0.4737499952316284, "rewards/judge_quality/std": 0.16291432082653046, "rewards/total_composite/mean": 0.6916118264198303, "rewards/total_composite/std": 0.15861403942108154, "reward": 0.6916118264198303, "reward_std": 0.15861402451992035, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19008465111255646, "sampling/sampling_logp_difference/max": 1.6687211990356445, "sampling/importance_sampling_ratio/min": 0.1884879618883133, "sampling/importance_sampling_ratio/mean": 1.0433640480041504, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8100511580705643, "clip_ratio/low_mean": 0.09510666877031326, "clip_ratio/low_min": 0.09510666877031326, "clip_ratio/high_mean": 0.09347247891128063, "clip_ratio/high_max": 0.09347247891128063, "clip_ratio/region_mean": 0.1885791476815939, "reward_total_mean": 0.6916118264198303, "reward_meter_mean": 0.7412514686584473, "reward_meter_std": 0.28295058012008667, "reward_count_adherence_mean": 0.7749999761581421, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9967364072799683, "reward_repeat_soft_std": 0.0038352853152900934, "reward_judge_quality_mean": 0.4737499952316284, "reward_judge_quality_std": 0.16291432082653046, "reward_total_composite_mean": 0.6916118264198303, "reward_total_composite_std": 0.15861403942108154} {"timestamp_utc": "2026-04-12T23:36:19Z", "mode": "train", "global_step": 572, "epoch": 0.057458563535911604, "loss": -0.1163, "grad_norm": 14.012733459472656, "learning_rate": 8.269696969696971e-06, "num_tokens": 1070557.0, "completions/mean_length": 40.625, "completions/min_length": 23.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.625, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.8592797517776489, "rewards/meter/std": 0.18029741942882538, "rewards/count_adherence/mean": 0.71875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9498857259750366, "rewards/repeat_soft/std": 0.04666195064783096, "rewards/judge_quality/mean": 0.5649999976158142, "rewards/judge_quality/std": 0.194054514169693, "rewards/total_composite/mean": 0.6707748174667358, "rewards/total_composite/std": 0.2873738408088684, "reward": 0.6707748174667358, "reward_std": 0.2873738408088684, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16684700548648834, "sampling/sampling_logp_difference/max": 2.452756881713867, "sampling/importance_sampling_ratio/min": 0.08605600893497467, "sampling/importance_sampling_ratio/mean": 0.9900557994842529, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0945413559675217, "clip_ratio/low_mean": 0.0241674380376935, "clip_ratio/low_min": 0.0241674380376935, "clip_ratio/high_mean": 0.1292927162721753, "clip_ratio/high_max": 0.1292927162721753, "clip_ratio/region_mean": 0.1534601543098688, "reward_total_mean": 0.6707748174667358, "reward_meter_mean": 0.8592797517776489, "reward_meter_std": 0.18029741942882538, "reward_count_adherence_mean": 0.71875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9498857259750366, "reward_repeat_soft_std": 0.04666195064783096, "reward_judge_quality_mean": 0.5649999976158142, "reward_judge_quality_std": 0.194054514169693, "reward_total_composite_mean": 0.6707748174667358, "reward_total_composite_std": 0.2873738408088684} {"timestamp_utc": "2026-04-12T23:36:26Z", "mode": "train", "global_step": 573, "epoch": 0.05755901557006529, "loss": 0.19, "grad_norm": 15.433377265930176, "learning_rate": 8.266666666666667e-06, "num_tokens": 1072391.0, "completions/mean_length": 50.25, "completions/min_length": 42.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.25, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9804360866546631, "rewards/meter/std": 0.012928040698170662, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.11785111576318741, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9869424104690552, "rewards/repeat_soft/std": 0.019997481256723404, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.6736487150192261, "rewards/total_composite/std": 0.2722592055797577, "reward": 0.6736487150192261, "reward_std": 0.2722591757774353, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16408930718898773, "sampling/sampling_logp_difference/max": 1.2016301155090332, "sampling/importance_sampling_ratio/min": 0.30070364475250244, "sampling/importance_sampling_ratio/mean": 1.0287965536117554, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5388877987861633, "clip_ratio/low_mean": 0.0211267601698637, "clip_ratio/low_min": 0.0211267601698637, "clip_ratio/high_mean": 0.14052575640380383, "clip_ratio/high_max": 0.14052575640380383, "clip_ratio/region_mean": 0.16165251657366753, "reward_total_mean": 0.6736487150192261, "reward_meter_mean": 0.9804360866546631, "reward_meter_std": 0.012928040698170662, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.11785111576318741, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9869424104690552, "reward_repeat_soft_std": 0.019997481256723404, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.6736487150192261, "reward_total_composite_std": 0.2722592055797577} {"timestamp_utc": "2026-04-12T23:36:32Z", "mode": "train", "global_step": 574, "epoch": 0.057659467604218986, "loss": 0.0567, "grad_norm": 24.0311279296875, "learning_rate": 8.263636363636366e-06, "num_tokens": 1073826.0, "completions/mean_length": 31.375, "completions/min_length": 28.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.375, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.49509674310684204, "rewards/meter/std": 0.302313894033432, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9894437789916992, "rewards/repeat_soft/std": 0.012283461168408394, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.6033629179000854, "rewards/total_composite/std": 0.13449133932590485, "reward": 0.6033629179000854, "reward_std": 0.13449135422706604, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19129028916358948, "sampling/sampling_logp_difference/max": 1.9234421253204346, "sampling/importance_sampling_ratio/min": 0.14610318839550018, "sampling/importance_sampling_ratio/mean": 1.0010403394699097, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8460233435034752, "clip_ratio/low_mean": 0.05087176733650267, "clip_ratio/low_min": 0.05087176733650267, "clip_ratio/high_mean": 0.06830357201397419, "clip_ratio/high_max": 0.06830357201397419, "clip_ratio/region_mean": 0.11917533935047686, "reward_total_mean": 0.6033629179000854, "reward_meter_mean": 0.49509674310684204, "reward_meter_std": 0.302313894033432, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9894437789916992, "reward_repeat_soft_std": 0.012283461168408394, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.6033629179000854, "reward_total_composite_std": 0.13449133932590485} {"timestamp_utc": "2026-04-12T23:36:38Z", "mode": "train", "global_step": 575, "epoch": 0.057759919638372674, "loss": 0.1195, "grad_norm": 18.55357551574707, "learning_rate": 8.260606060606061e-06, "num_tokens": 1075488.0, "completions/mean_length": 37.75, "completions/min_length": 28.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.75, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.6509860754013062, "rewards/meter/std": 0.416098952293396, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9751641750335693, "rewards/repeat_soft/std": 0.040795374661684036, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.6709601879119873, "rewards/total_composite/std": 0.18830597400665283, "reward": 0.6709601879119873, "reward_std": 0.18830597400665283, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1665351837873459, "sampling/sampling_logp_difference/max": 3.833083152770996, "sampling/importance_sampling_ratio/min": 0.021642785519361496, "sampling/importance_sampling_ratio/mean": 1.022983431816101, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3059828728437424, "clip_ratio/low_mean": 0.05400980915874243, "clip_ratio/low_min": 0.05400980915874243, "clip_ratio/high_mean": 0.09693451318889856, "clip_ratio/high_max": 0.09693451318889856, "clip_ratio/region_mean": 0.150944322347641, "reward_total_mean": 0.6709601879119873, "reward_meter_mean": 0.6509860754013062, "reward_meter_std": 0.416098952293396, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9751641750335693, "reward_repeat_soft_std": 0.040795374661684036, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.6709601879119873, "reward_total_composite_std": 0.18830597400665283} {"timestamp_utc": "2026-04-12T23:36:44Z", "mode": "train", "global_step": 576, "epoch": 0.05786037167252637, "loss": 0.0989, "grad_norm": 34.63580322265625, "learning_rate": 8.257575757575758e-06, "num_tokens": 1076908.0, "completions/mean_length": 26.5, "completions/min_length": 23.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.5, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.4068511128425598, "rewards/meter/std": 0.3305554687976837, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9940686821937561, "rewards/repeat_soft/std": 0.009038038551807404, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.4701748490333557, "rewards/total_composite/std": 0.22988100349903107, "reward": 0.4701748490333557, "reward_std": 0.22988098859786987, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2147468775510788, "sampling/sampling_logp_difference/max": 1.9677866697311401, "sampling/importance_sampling_ratio/min": 0.13976585865020752, "sampling/importance_sampling_ratio/mean": 0.9918428063392639, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1101408749818802, "clip_ratio/low_mean": 0.12498558312654495, "clip_ratio/low_min": 0.12498558312654495, "clip_ratio/high_mean": 0.05501706013455987, "clip_ratio/high_max": 0.05501706013455987, "clip_ratio/region_mean": 0.18000264326110482, "reward_total_mean": 0.4701748490333557, "reward_meter_mean": 0.4068511128425598, "reward_meter_std": 0.3305554687976837, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9940686821937561, "reward_repeat_soft_std": 0.009038038551807404, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.4701748490333557, "reward_total_composite_std": 0.22988100349903107} {"timestamp_utc": "2026-04-12T23:36:51Z", "mode": "train", "global_step": 577, "epoch": 0.05796082370668006, "loss": 0.0345, "grad_norm": 13.662009239196777, "learning_rate": 8.254545454545456e-06, "num_tokens": 1078931.0, "completions/mean_length": 61.875, "completions/min_length": 57.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.875, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.8055605888366699, "rewards/meter/std": 0.22162972390651703, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9780954122543335, "rewards/repeat_soft/std": 0.025631239637732506, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1240895539522171, "rewards/total_composite/mean": 0.7186868190765381, "rewards/total_composite/std": 0.10872305929660797, "reward": 0.7186868190765381, "reward_std": 0.10872308909893036, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1810440868139267, "sampling/sampling_logp_difference/max": 2.281038284301758, "sampling/importance_sampling_ratio/min": 0.10217805951833725, "sampling/importance_sampling_ratio/mean": 1.0164777040481567, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4337783306837082, "clip_ratio/low_mean": 0.03510845359414816, "clip_ratio/low_min": 0.03510845359414816, "clip_ratio/high_mean": 0.11550858523696661, "clip_ratio/high_max": 0.11550858523696661, "clip_ratio/region_mean": 0.15061703883111477, "reward_total_mean": 0.7186868190765381, "reward_meter_mean": 0.8055605888366699, "reward_meter_std": 0.22162972390651703, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9780954122543335, "reward_repeat_soft_std": 0.025631239637732506, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1240895539522171, "reward_total_composite_mean": 0.7186868190765381, "reward_total_composite_std": 0.10872305929660797} {"timestamp_utc": "2026-04-12T23:36:57Z", "mode": "train", "global_step": 578, "epoch": 0.05806127574083375, "loss": 0.0499, "grad_norm": 13.301177024841309, "learning_rate": 8.251515151515153e-06, "num_tokens": 1080549.0, "completions/mean_length": 45.25, "completions/min_length": 40.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.25, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9835100173950195, "rewards/meter/std": 0.016741542145609856, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.98753821849823, "rewards/repeat_soft/std": 0.01584974303841591, "rewards/judge_quality/mean": 0.6725000143051147, "rewards/judge_quality/std": 0.2068643569946289, "rewards/total_composite/mean": 0.8930832743644714, "rewards/total_composite/std": 0.06450106203556061, "reward": 0.8930832743644714, "reward_std": 0.06450106203556061, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18133074045181274, "sampling/sampling_logp_difference/max": 1.365549087524414, "sampling/importance_sampling_ratio/min": 0.2552404999732971, "sampling/importance_sampling_ratio/mean": 1.025810956954956, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4027932584285736, "clip_ratio/low_mean": 0.05594315379858017, "clip_ratio/low_min": 0.05594315379858017, "clip_ratio/high_mean": 0.09514554683119059, "clip_ratio/high_max": 0.09514554683119059, "clip_ratio/region_mean": 0.15108870062977076, "reward_total_mean": 0.8930832743644714, "reward_meter_mean": 0.9835100173950195, "reward_meter_std": 0.016741542145609856, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.98753821849823, "reward_repeat_soft_std": 0.01584974303841591, "reward_judge_quality_mean": 0.6725000143051147, "reward_judge_quality_std": 0.2068643569946289, "reward_total_composite_mean": 0.8930832743644714, "reward_total_composite_std": 0.06450106203556061} {"timestamp_utc": "2026-04-12T23:37:03Z", "mode": "train", "global_step": 579, "epoch": 0.058161727774987446, "loss": 0.0227, "grad_norm": 14.055227279663086, "learning_rate": 8.248484848484848e-06, "num_tokens": 1082151.0, "completions/mean_length": 44.25, "completions/min_length": 37.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.25, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.5616954565048218, "rewards/meter/std": 0.37994927167892456, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9984158873558044, "rewards/repeat_soft/std": 0.0021416647359728813, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.7058545351028442, "rewards/total_composite/std": 0.1558811217546463, "reward": 0.7058545351028442, "reward_std": 0.1558811217546463, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21068182587623596, "sampling/sampling_logp_difference/max": 3.6113224029541016, "sampling/importance_sampling_ratio/min": 0.02701609767973423, "sampling/importance_sampling_ratio/mean": 1.0284401178359985, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7849018573760986, "clip_ratio/low_mean": 0.10791454464197159, "clip_ratio/low_min": 0.10791454464197159, "clip_ratio/high_mean": 0.07566041126847267, "clip_ratio/high_max": 0.07566041126847267, "clip_ratio/region_mean": 0.18357495591044426, "reward_total_mean": 0.7058545351028442, "reward_meter_mean": 0.5616954565048218, "reward_meter_std": 0.37994927167892456, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9984158873558044, "reward_repeat_soft_std": 0.0021416647359728813, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.7058545351028442, "reward_total_composite_std": 0.1558811217546463} {"timestamp_utc": "2026-04-12T23:37:09Z", "mode": "train", "global_step": 580, "epoch": 0.05826217980914113, "loss": 0.0445, "grad_norm": 13.066967010498047, "learning_rate": 8.245454545454546e-06, "num_tokens": 1084078.0, "completions/mean_length": 61.875, "completions/min_length": 42.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.875, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.8593478202819824, "rewards/meter/std": 0.25535938143730164, "rewards/count_adherence/mean": 0.71875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9929438829421997, "rewards/repeat_soft/std": 0.007092205807566643, "rewards/judge_quality/mean": 0.5737500190734863, "rewards/judge_quality/std": 0.16465875506401062, "rewards/total_composite/mean": 0.765938401222229, "rewards/total_composite/std": 0.13207170367240906, "reward": 0.765938401222229, "reward_std": 0.13207168877124786, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17147260904312134, "sampling/sampling_logp_difference/max": 2.8095650672912598, "sampling/importance_sampling_ratio/min": 0.06023118272423744, "sampling/importance_sampling_ratio/mean": 1.000364065170288, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3274941816926003, "clip_ratio/low_mean": 0.03486717212945223, "clip_ratio/low_min": 0.03486717212945223, "clip_ratio/high_mean": 0.11801252886652946, "clip_ratio/high_max": 0.11801252886652946, "clip_ratio/region_mean": 0.1528797009959817, "reward_total_mean": 0.765938401222229, "reward_meter_mean": 0.8593478202819824, "reward_meter_std": 0.25535938143730164, "reward_count_adherence_mean": 0.71875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9929438829421997, "reward_repeat_soft_std": 0.007092205807566643, "reward_judge_quality_mean": 0.5737500190734863, "reward_judge_quality_std": 0.16465875506401062, "reward_total_composite_mean": 0.765938401222229, "reward_total_composite_std": 0.13207170367240906} {"timestamp_utc": "2026-04-12T23:37:16Z", "mode": "train", "global_step": 581, "epoch": 0.05836263184329483, "loss": 0.0358, "grad_norm": 18.46389389038086, "learning_rate": 8.242424242424243e-06, "num_tokens": 1085850.0, "completions/mean_length": 39.5, "completions/min_length": 33.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.5, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9742410778999329, "rewards/meter/std": 0.014273247681558132, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.11785111576318741, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9963656663894653, "rewards/repeat_soft/std": 0.0042617591097950935, "rewards/judge_quality/mean": 0.7112500071525574, "rewards/judge_quality/std": 0.2928401827812195, "rewards/total_composite/mean": 0.8576700687408447, "rewards/total_composite/std": 0.07859523594379425, "reward": 0.8576700687408447, "reward_std": 0.07859524339437485, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19024407863616943, "sampling/sampling_logp_difference/max": 1.6292905807495117, "sampling/importance_sampling_ratio/min": 0.19606861472129822, "sampling/importance_sampling_ratio/mean": 1.0171968936920166, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3431769013404846, "clip_ratio/low_mean": 0.0535316402092576, "clip_ratio/low_min": 0.0535316402092576, "clip_ratio/high_mean": 0.09818596672266722, "clip_ratio/high_max": 0.09818596672266722, "clip_ratio/region_mean": 0.15171760693192482, "reward_total_mean": 0.8576700687408447, "reward_meter_mean": 0.9742410778999329, "reward_meter_std": 0.014273247681558132, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.11785111576318741, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9963656663894653, "reward_repeat_soft_std": 0.0042617591097950935, "reward_judge_quality_mean": 0.7112500071525574, "reward_judge_quality_std": 0.2928401827812195, "reward_total_composite_mean": 0.8576700687408447, "reward_total_composite_std": 0.07859523594379425} {"timestamp_utc": "2026-04-12T23:37:22Z", "mode": "train", "global_step": 582, "epoch": 0.058463083877448516, "loss": 0.0839, "grad_norm": 19.778390884399414, "learning_rate": 8.23939393939394e-06, "num_tokens": 1087625.0, "completions/mean_length": 46.875, "completions/min_length": 41.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.875, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.6848108768463135, "rewards/meter/std": 0.30949491262435913, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9806612730026245, "rewards/repeat_soft/std": 0.01394400279968977, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1240895539522171, "rewards/total_composite/mean": 0.6646060347557068, "rewards/total_composite/std": 0.12235091626644135, "reward": 0.6646060347557068, "reward_std": 0.12235090881586075, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17743471264839172, "sampling/sampling_logp_difference/max": 2.6238396167755127, "sampling/importance_sampling_ratio/min": 0.07252386957406998, "sampling/importance_sampling_ratio/mean": 1.0283068418502808, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9549873396754265, "clip_ratio/low_mean": 0.048958334140479565, "clip_ratio/low_min": 0.048958334140479565, "clip_ratio/high_mean": 0.09031765535473824, "clip_ratio/high_max": 0.09031765535473824, "clip_ratio/region_mean": 0.1392759894952178, "reward_total_mean": 0.6646060347557068, "reward_meter_mean": 0.6848108768463135, "reward_meter_std": 0.30949491262435913, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9806612730026245, "reward_repeat_soft_std": 0.01394400279968977, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1240895539522171, "reward_total_composite_mean": 0.6646060347557068, "reward_total_composite_std": 0.12235091626644135} {"timestamp_utc": "2026-04-12T23:37:29Z", "mode": "train", "global_step": 583, "epoch": 0.05856353591160221, "loss": -0.0489, "grad_norm": 14.818984985351562, "learning_rate": 8.236363636363637e-06, "num_tokens": 1089191.0, "completions/mean_length": 37.75, "completions/min_length": 30.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.75, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.7802342176437378, "rewards/meter/std": 0.3672329783439636, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9909406304359436, "rewards/repeat_soft/std": 0.011600243858993053, "rewards/judge_quality/mean": 0.4649999737739563, "rewards/judge_quality/std": 0.11563490331172943, "rewards/total_composite/mean": 0.7396994829177856, "rewards/total_composite/std": 0.14878863096237183, "reward": 0.7396994829177856, "reward_std": 0.14878863096237183, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18382035195827484, "sampling/sampling_logp_difference/max": 1.5794780254364014, "sampling/importance_sampling_ratio/min": 0.20608264207839966, "sampling/importance_sampling_ratio/mean": 1.0195516347885132, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5667392164468765, "clip_ratio/low_mean": 0.039880952797830105, "clip_ratio/low_min": 0.039880952797830105, "clip_ratio/high_mean": 0.13365895207971334, "clip_ratio/high_max": 0.13365895207971334, "clip_ratio/region_mean": 0.17353990487754345, "reward_total_mean": 0.7396994829177856, "reward_meter_mean": 0.7802342176437378, "reward_meter_std": 0.3672329783439636, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9909406304359436, "reward_repeat_soft_std": 0.011600243858993053, "reward_judge_quality_mean": 0.4649999737739563, "reward_judge_quality_std": 0.11563490331172943, "reward_total_composite_mean": 0.7396994829177856, "reward_total_composite_std": 0.14878863096237183} {"timestamp_utc": "2026-04-12T23:37:36Z", "mode": "train", "global_step": 584, "epoch": 0.058663987945755905, "loss": 0.0278, "grad_norm": 8.632674217224121, "learning_rate": 8.233333333333335e-06, "num_tokens": 1091381.0, "completions/mean_length": 93.75, "completions/min_length": 84.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.75, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.950150728225708, "rewards/meter/std": 0.08896130323410034, "rewards/count_adherence/mean": 0.7749999761581421, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9713970422744751, "rewards/repeat_soft/std": 0.0441599041223526, "rewards/judge_quality/mean": 0.5112500190734863, "rewards/judge_quality/std": 0.18216457962989807, "rewards/total_composite/mean": 0.7051737308502197, "rewards/total_composite/std": 0.2887442409992218, "reward": 0.7051737308502197, "reward_std": 0.2887442409992218, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17510424554347992, "sampling/sampling_logp_difference/max": 1.5146093368530273, "sampling/importance_sampling_ratio/min": 0.21989408135414124, "sampling/importance_sampling_ratio/mean": 1.0074076652526855, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7748349756002426, "clip_ratio/low_mean": 0.010416666977107525, "clip_ratio/low_min": 0.010416666977107525, "clip_ratio/high_mean": 0.16546723060309887, "clip_ratio/high_max": 0.16546723060309887, "clip_ratio/region_mean": 0.1758838975802064, "reward_total_mean": 0.7051737308502197, "reward_meter_mean": 0.950150728225708, "reward_meter_std": 0.08896130323410034, "reward_count_adherence_mean": 0.7749999761581421, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9713970422744751, "reward_repeat_soft_std": 0.0441599041223526, "reward_judge_quality_mean": 0.5112500190734863, "reward_judge_quality_std": 0.18216457962989807, "reward_total_composite_mean": 0.7051737308502197, "reward_total_composite_std": 0.2887442409992218} {"timestamp_utc": "2026-04-12T23:37:42Z", "mode": "train", "global_step": 585, "epoch": 0.05876443997990959, "loss": 0.0165, "grad_norm": 24.147626876831055, "learning_rate": 8.23030303030303e-06, "num_tokens": 1092766.0, "completions/mean_length": 22.125, "completions/min_length": 21.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.125, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.685529351234436, "rewards/meter/std": 0.3062390089035034, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.19334925711154938, "rewards/total_composite/mean": 0.5995575785636902, "rewards/total_composite/std": 0.28815168142318726, "reward": 0.5995575785636902, "reward_std": 0.28815165162086487, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11658532917499542, "sampling/sampling_logp_difference/max": 1.377192497253418, "sampling/importance_sampling_ratio/min": 0.2522858679294586, "sampling/importance_sampling_ratio/mean": 1.0046762228012085, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.866596058011055, "clip_ratio/low_mean": 0.02922077989205718, "clip_ratio/low_min": 0.02922077989205718, "clip_ratio/high_mean": 0.04990118695423007, "clip_ratio/high_max": 0.04990118695423007, "clip_ratio/region_mean": 0.07912196684628725, "reward_total_mean": 0.5995575785636902, "reward_meter_mean": 0.685529351234436, "reward_meter_std": 0.3062390089035034, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.19334925711154938, "reward_total_composite_mean": 0.5995575785636902, "reward_total_composite_std": 0.28815168142318726} {"timestamp_utc": "2026-04-12T23:37:49Z", "mode": "train", "global_step": 586, "epoch": 0.05886489201406329, "loss": -0.0273, "grad_norm": 14.518970489501953, "learning_rate": 8.227272727272728e-06, "num_tokens": 1094541.0, "completions/mean_length": 60.875, "completions/min_length": 48.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.8596857786178589, "rewards/meter/std": 0.32018041610717773, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9894834756851196, "rewards/repeat_soft/std": 0.0106501504778862, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7243069410324097, "rewards/total_composite/std": 0.14370694756507874, "reward": 0.7243069410324097, "reward_std": 0.14370694756507874, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1730169951915741, "sampling/sampling_logp_difference/max": 1.9252433776855469, "sampling/importance_sampling_ratio/min": 0.14584025740623474, "sampling/importance_sampling_ratio/mean": 1.021816372871399, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.454993337392807, "clip_ratio/low_mean": 0.015625, "clip_ratio/low_min": 0.015625, "clip_ratio/high_mean": 0.15178421512246132, "clip_ratio/high_max": 0.15178421512246132, "clip_ratio/region_mean": 0.16740921512246132, "reward_total_mean": 0.7243069410324097, "reward_meter_mean": 0.8596857786178589, "reward_meter_std": 0.32018041610717773, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9894834756851196, "reward_repeat_soft_std": 0.0106501504778862, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7243069410324097, "reward_total_composite_std": 0.14370694756507874} {"timestamp_utc": "2026-04-12T23:37:56Z", "mode": "train", "global_step": 587, "epoch": 0.058965344048216975, "loss": -0.0129, "grad_norm": 20.017915725708008, "learning_rate": 8.224242424242425e-06, "num_tokens": 1095902.0, "completions/mean_length": 21.125, "completions/min_length": 17.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.125, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.8852671384811401, "rewards/meter/std": 0.13455387949943542, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9465909004211426, "rewards/repeat_soft/std": 0.04499770700931549, "rewards/judge_quality/mean": 0.44624999165534973, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.776904284954071, "rewards/total_composite/std": 0.059851281344890594, "reward": 0.776904284954071, "reward_std": 0.0598512701690197, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17502668499946594, "sampling/sampling_logp_difference/max": 1.070300579071045, "sampling/importance_sampling_ratio/min": 0.4083203077316284, "sampling/importance_sampling_ratio/mean": 1.0317704677581787, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7006223052740097, "clip_ratio/low_mean": 0.08640350960195065, "clip_ratio/low_min": 0.08640350960195065, "clip_ratio/high_mean": 0.09314800705760717, "clip_ratio/high_max": 0.09314800705760717, "clip_ratio/region_mean": 0.17955151665955782, "reward_total_mean": 0.776904284954071, "reward_meter_mean": 0.8852671384811401, "reward_meter_std": 0.13455387949943542, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9465909004211426, "reward_repeat_soft_std": 0.04499770700931549, "reward_judge_quality_mean": 0.44624999165534973, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.776904284954071, "reward_total_composite_std": 0.059851281344890594} {"timestamp_utc": "2026-04-12T23:38:03Z", "mode": "train", "global_step": 588, "epoch": 0.05906579608237067, "loss": -0.0077, "grad_norm": 15.826994895935059, "learning_rate": 8.221212121212122e-06, "num_tokens": 1097500.0, "completions/mean_length": 38.75, "completions/min_length": 27.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.75, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.48526209592819214, "rewards/meter/std": 0.37712329626083374, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9899910688400269, "rewards/repeat_soft/std": 0.011189603246748447, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.6706170439720154, "rewards/total_composite/std": 0.17903757095336914, "reward": 0.6706170439720154, "reward_std": 0.17903757095336914, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16439051926136017, "sampling/sampling_logp_difference/max": 1.105130672454834, "sampling/importance_sampling_ratio/min": 0.331167608499527, "sampling/importance_sampling_ratio/mean": 1.0379860401153564, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3430720940232277, "clip_ratio/low_mean": 0.08696564752608538, "clip_ratio/low_min": 0.08696564752608538, "clip_ratio/high_mean": 0.0722828283905983, "clip_ratio/high_max": 0.0722828283905983, "clip_ratio/region_mean": 0.15924847591668367, "reward_total_mean": 0.6706170439720154, "reward_meter_mean": 0.48526209592819214, "reward_meter_std": 0.37712329626083374, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9899910688400269, "reward_repeat_soft_std": 0.011189603246748447, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.6706170439720154, "reward_total_composite_std": 0.17903757095336914} {"timestamp_utc": "2026-04-12T23:38:09Z", "mode": "train", "global_step": 589, "epoch": 0.05916624811652436, "loss": 0.0107, "grad_norm": 45.62667465209961, "learning_rate": 8.21818181818182e-06, "num_tokens": 1098852.0, "completions/mean_length": 22.0, "completions/min_length": 19.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.0, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.5676230192184448, "rewards/meter/std": 0.45576298236846924, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9555155038833618, "rewards/repeat_soft/std": 0.013097990304231644, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.704231858253479, "rewards/total_composite/std": 0.2306322604417801, "reward": 0.704231858253479, "reward_std": 0.2306322604417801, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15652644634246826, "sampling/sampling_logp_difference/max": 1.6905694007873535, "sampling/importance_sampling_ratio/min": 0.18441449105739594, "sampling/importance_sampling_ratio/mean": 0.9911974668502808, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1740966811776161, "clip_ratio/low_mean": 0.03538329526782036, "clip_ratio/low_min": 0.03538329526782036, "clip_ratio/high_mean": 0.08109887409955263, "clip_ratio/high_max": 0.08109887409955263, "clip_ratio/region_mean": 0.11648216936737299, "reward_total_mean": 0.704231858253479, "reward_meter_mean": 0.5676230192184448, "reward_meter_std": 0.45576298236846924, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9555155038833618, "reward_repeat_soft_std": 0.013097990304231644, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.704231858253479, "reward_total_composite_std": 0.2306322604417801} {"timestamp_utc": "2026-04-12T23:38:17Z", "mode": "train", "global_step": 590, "epoch": 0.05926670015067805, "loss": 0.0754, "grad_norm": 18.590423583984375, "learning_rate": 8.215151515151517e-06, "num_tokens": 1100252.0, "completions/mean_length": 22.0, "completions/min_length": 17.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.0, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9912739396095276, "rewards/meter/std": 0.00444032484665513, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9220082759857178, "rewards/repeat_soft/std": 0.11452778428792953, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.8352741003036499, "rewards/total_composite/std": 0.055369120091199875, "reward": 0.8352741003036499, "reward_std": 0.05536911264061928, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14577960968017578, "sampling/sampling_logp_difference/max": 1.0432777404785156, "sampling/importance_sampling_ratio/min": 0.3522980511188507, "sampling/importance_sampling_ratio/mean": 1.0369802713394165, "sampling/importance_sampling_ratio/max": 1.7174524068832397, "entropy": 1.719233438372612, "clip_ratio/low_mean": 0.13486368767917156, "clip_ratio/low_min": 0.13486368767917156, "clip_ratio/high_mean": 0.017045455053448677, "clip_ratio/high_max": 0.017045455053448677, "clip_ratio/region_mean": 0.15190914273262024, "reward_total_mean": 0.8352741003036499, "reward_meter_mean": 0.9912739396095276, "reward_meter_std": 0.00444032484665513, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9220082759857178, "reward_repeat_soft_std": 0.11452778428792953, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.8352741003036499, "reward_total_composite_std": 0.055369120091199875} {"timestamp_utc": "2026-04-12T23:38:23Z", "mode": "train", "global_step": 591, "epoch": 0.05936715218483174, "loss": -0.0578, "grad_norm": 18.889339447021484, "learning_rate": 8.212121212121212e-06, "num_tokens": 1102037.0, "completions/mean_length": 45.125, "completions/min_length": 32.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.125, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9563775062561035, "rewards/meter/std": 0.022522713989019394, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.17251639068126678, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9887794852256775, "rewards/repeat_soft/std": 0.014271514490246773, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.8989978432655334, "rewards/total_composite/std": 0.08217870444059372, "reward": 0.8989978432655334, "reward_std": 0.08217868953943253, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16878581047058105, "sampling/sampling_logp_difference/max": 2.2135801315307617, "sampling/importance_sampling_ratio/min": 0.10930860787630081, "sampling/importance_sampling_ratio/mean": 1.0167531967163086, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4341708347201347, "clip_ratio/low_mean": 0.04673763830214739, "clip_ratio/low_min": 0.04673763830214739, "clip_ratio/high_mean": 0.10145634412765503, "clip_ratio/high_max": 0.10145634412765503, "clip_ratio/region_mean": 0.14819398242980242, "reward_total_mean": 0.8989978432655334, "reward_meter_mean": 0.9563775062561035, "reward_meter_std": 0.022522713989019394, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.17251639068126678, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9887794852256775, "reward_repeat_soft_std": 0.014271514490246773, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.8989978432655334, "reward_total_composite_std": 0.08217870444059372} {"timestamp_utc": "2026-04-12T23:38:35Z", "mode": "train", "global_step": 592, "epoch": 0.059467604218985434, "loss": -0.1472, "grad_norm": 2.1594531536102295, "learning_rate": 8.20909090909091e-06, "num_tokens": 1103834.0, "completions/mean_length": 239.625, "completions/min_length": 69.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 76.20000457763672, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.8745372295379639, "rewards/meter/std": 0.17252445220947266, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9981652498245239, "rewards/repeat_soft/std": 0.00317987147718668, "rewards/judge_quality/mean": 0.29750001430511475, "rewards/judge_quality/std": 0.24188250303268433, "rewards/total_composite/mean": 0.48399385809898376, "rewards/total_composite/std": 0.4010038375854492, "reward": 0.48399385809898376, "reward_std": 0.40100380778312683, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.191337451338768, "sampling/sampling_logp_difference/max": 1.1521296501159668, "sampling/importance_sampling_ratio/min": 0.31596314907073975, "sampling/importance_sampling_ratio/mean": 1.0628008842468262, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.72052863240242, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09722495172172785, "clip_ratio/high_max": 0.09722495172172785, "clip_ratio/region_mean": 0.09722495172172785, "reward_total_mean": 0.48399385809898376, "reward_meter_mean": 0.8745372295379639, "reward_meter_std": 0.17252445220947266, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9981652498245239, "reward_repeat_soft_std": 0.00317987147718668, "reward_judge_quality_mean": 0.29750001430511475, "reward_judge_quality_std": 0.24188250303268433, "reward_total_composite_mean": 0.48399385809898376, "reward_total_composite_std": 0.4010038375854492} {"timestamp_utc": "2026-04-12T23:38:42Z", "mode": "train", "global_step": 593, "epoch": 0.05956805625313913, "loss": 0.0749, "grad_norm": 19.545875549316406, "learning_rate": 8.206060606060607e-06, "num_tokens": 1105587.0, "completions/mean_length": 58.125, "completions/min_length": 44.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.125, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.8536458015441895, "rewards/meter/std": 0.33986762166023254, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9956611394882202, "rewards/repeat_soft/std": 0.006612555123865604, "rewards/judge_quality/mean": 0.5687500238418579, "rewards/judge_quality/std": 0.19111984968185425, "rewards/total_composite/mean": 0.6687114238739014, "rewards/total_composite/std": 0.2949616611003876, "reward": 0.6687114238739014, "reward_std": 0.2949616611003876, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18272416293621063, "sampling/sampling_logp_difference/max": 1.738306999206543, "sampling/importance_sampling_ratio/min": 0.1758178025484085, "sampling/importance_sampling_ratio/mean": 1.0262454748153687, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7533404380083084, "clip_ratio/low_mean": 0.045085471123456955, "clip_ratio/low_min": 0.045085471123456955, "clip_ratio/high_mean": 0.1413087546825409, "clip_ratio/high_max": 0.1413087546825409, "clip_ratio/region_mean": 0.18639422580599785, "reward_total_mean": 0.6687114238739014, "reward_meter_mean": 0.8536458015441895, "reward_meter_std": 0.33986762166023254, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9956611394882202, "reward_repeat_soft_std": 0.006612555123865604, "reward_judge_quality_mean": 0.5687500238418579, "reward_judge_quality_std": 0.19111984968185425, "reward_total_composite_mean": 0.6687114238739014, "reward_total_composite_std": 0.2949616611003876} {"timestamp_utc": "2026-04-12T23:38:49Z", "mode": "train", "global_step": 594, "epoch": 0.05966850828729282, "loss": 0.0216, "grad_norm": 14.390789985656738, "learning_rate": 8.203030303030304e-06, "num_tokens": 1107286.0, "completions/mean_length": 49.375, "completions/min_length": 42.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.375, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.4760621190071106, "rewards/meter/std": 0.35368868708610535, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9919086694717407, "rewards/repeat_soft/std": 0.008062063716351986, "rewards/judge_quality/mean": 0.6200000047683716, "rewards/judge_quality/std": 0.22677870094776154, "rewards/total_composite/mean": 0.6119188070297241, "rewards/total_composite/std": 0.1924046277999878, "reward": 0.6119188070297241, "reward_std": 0.1924046277999878, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15076735615730286, "sampling/sampling_logp_difference/max": 1.1296372413635254, "sampling/importance_sampling_ratio/min": 0.3231504559516907, "sampling/importance_sampling_ratio/mean": 1.0281649827957153, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9687571078538895, "clip_ratio/low_mean": 0.08337779622524977, "clip_ratio/low_min": 0.08337779622524977, "clip_ratio/high_mean": 0.059571911580860615, "clip_ratio/high_max": 0.059571911580860615, "clip_ratio/region_mean": 0.14294970780611038, "reward_total_mean": 0.6119188070297241, "reward_meter_mean": 0.4760621190071106, "reward_meter_std": 0.35368868708610535, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9919086694717407, "reward_repeat_soft_std": 0.008062063716351986, "reward_judge_quality_mean": 0.6200000047683716, "reward_judge_quality_std": 0.22677870094776154, "reward_total_composite_mean": 0.6119188070297241, "reward_total_composite_std": 0.1924046277999878} {"timestamp_utc": "2026-04-12T23:38:56Z", "mode": "train", "global_step": 595, "epoch": 0.05976896032144651, "loss": 0.0274, "grad_norm": 19.314571380615234, "learning_rate": 8.2e-06, "num_tokens": 1108686.0, "completions/mean_length": 25.0, "completions/min_length": 19.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.0, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.5756844282150269, "rewards/meter/std": 0.4438043534755707, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9569318294525146, "rewards/repeat_soft/std": 0.015749182552099228, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.6506261825561523, "rewards/total_composite/std": 0.22602246701717377, "reward": 0.6506261825561523, "reward_std": 0.22602245211601257, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1889304369688034, "sampling/sampling_logp_difference/max": 1.1023235321044922, "sampling/importance_sampling_ratio/min": 0.3320985436439514, "sampling/importance_sampling_ratio/mean": 1.0343499183654785, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8472767621278763, "clip_ratio/low_mean": 0.06340225413441658, "clip_ratio/low_min": 0.06340225413441658, "clip_ratio/high_mean": 0.09621279826387763, "clip_ratio/high_max": 0.09621279826387763, "clip_ratio/region_mean": 0.1596150523982942, "reward_total_mean": 0.6506261825561523, "reward_meter_mean": 0.5756844282150269, "reward_meter_std": 0.4438043534755707, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9569318294525146, "reward_repeat_soft_std": 0.015749182552099228, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.6506261825561523, "reward_total_composite_std": 0.22602246701717377} {"timestamp_utc": "2026-04-12T23:39:03Z", "mode": "train", "global_step": 596, "epoch": 0.0598694123556002, "loss": 0.1747, "grad_norm": 13.732975959777832, "learning_rate": 8.196969696969698e-06, "num_tokens": 1110576.0, "completions/mean_length": 61.25, "completions/min_length": 41.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.25, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.5563759207725525, "rewards/meter/std": 0.26193952560424805, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9973032474517822, "rewards/repeat_soft/std": 0.001868619117885828, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.0843462198972702, "rewards/total_composite/mean": 0.592162013053894, "rewards/total_composite/std": 0.1313713937997818, "reward": 0.592162013053894, "reward_std": 0.131371408700943, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18713976442813873, "sampling/sampling_logp_difference/max": 1.8935699462890625, "sampling/importance_sampling_ratio/min": 0.15053345263004303, "sampling/importance_sampling_ratio/mean": 1.0374755859375, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8350261226296425, "clip_ratio/low_mean": 0.08667160756886005, "clip_ratio/low_min": 0.08667160756886005, "clip_ratio/high_mean": 0.08797039557248354, "clip_ratio/high_max": 0.08797039557248354, "clip_ratio/region_mean": 0.1746420031413436, "reward_total_mean": 0.592162013053894, "reward_meter_mean": 0.5563759207725525, "reward_meter_std": 0.26193952560424805, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9973032474517822, "reward_repeat_soft_std": 0.001868619117885828, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.0843462198972702, "reward_total_composite_mean": 0.592162013053894, "reward_total_composite_std": 0.1313713937997818} {"timestamp_utc": "2026-04-12T23:39:10Z", "mode": "train", "global_step": 597, "epoch": 0.059969864389753894, "loss": 0.0453, "grad_norm": 9.9744873046875, "learning_rate": 8.193939393939394e-06, "num_tokens": 1112500.0, "completions/mean_length": 69.5, "completions/min_length": 61.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.5, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.982949435710907, "rewards/meter/std": 0.01799902506172657, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9968470335006714, "rewards/repeat_soft/std": 0.0043213097378611565, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.21084439754486084, "rewards/total_composite/mean": 0.815761923789978, "rewards/total_composite/std": 0.054767899215221405, "reward": 0.815761923789978, "reward_std": 0.054767902940511703, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20237718522548676, "sampling/sampling_logp_difference/max": 1.3700478076934814, "sampling/importance_sampling_ratio/min": 0.25409480929374695, "sampling/importance_sampling_ratio/mean": 1.0584841966629028, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0005133897066116, "clip_ratio/low_mean": 0.10281671117991209, "clip_ratio/low_min": 0.10281671117991209, "clip_ratio/high_mean": 0.06646348536014557, "clip_ratio/high_max": 0.06646348536014557, "clip_ratio/region_mean": 0.16928019654005766, "reward_total_mean": 0.815761923789978, "reward_meter_mean": 0.982949435710907, "reward_meter_std": 0.01799902506172657, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9968470335006714, "reward_repeat_soft_std": 0.0043213097378611565, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.21084439754486084, "reward_total_composite_mean": 0.815761923789978, "reward_total_composite_std": 0.054767899215221405} {"timestamp_utc": "2026-04-12T23:39:16Z", "mode": "train", "global_step": 598, "epoch": 0.06007031642390758, "loss": 0.0307, "grad_norm": 15.408626556396484, "learning_rate": 8.190909090909091e-06, "num_tokens": 1114063.0, "completions/mean_length": 39.375, "completions/min_length": 34.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.375, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.6237105131149292, "rewards/meter/std": 0.4237210154533386, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9853572845458984, "rewards/repeat_soft/std": 0.023067643865942955, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.677330493927002, "rewards/total_composite/std": 0.20981962978839874, "reward": 0.677330493927002, "reward_std": 0.20981964468955994, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1833791434764862, "sampling/sampling_logp_difference/max": 1.1559257507324219, "sampling/importance_sampling_ratio/min": 0.314765989780426, "sampling/importance_sampling_ratio/mean": 1.0348992347717285, "sampling/importance_sampling_ratio/max": 1.9759782552719116, "entropy": 1.971277967095375, "clip_ratio/low_mean": 0.0504638385027647, "clip_ratio/low_min": 0.0504638385027647, "clip_ratio/high_mean": 0.10691743157804012, "clip_ratio/high_max": 0.10691743157804012, "clip_ratio/region_mean": 0.15738127008080482, "reward_total_mean": 0.677330493927002, "reward_meter_mean": 0.6237105131149292, "reward_meter_std": 0.4237210154533386, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9853572845458984, "reward_repeat_soft_std": 0.023067643865942955, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.677330493927002, "reward_total_composite_std": 0.20981962978839874} {"timestamp_utc": "2026-04-12T23:39:23Z", "mode": "train", "global_step": 599, "epoch": 0.060170768458061276, "loss": 0.0817, "grad_norm": 14.309142112731934, "learning_rate": 8.187878787878788e-06, "num_tokens": 1115753.0, "completions/mean_length": 41.25, "completions/min_length": 33.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.25, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.4323441982269287, "rewards/meter/std": 0.4019242823123932, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9930334091186523, "rewards/repeat_soft/std": 0.010946320369839668, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5732332468032837, "rewards/total_composite/std": 0.18157166242599487, "reward": 0.5732332468032837, "reward_std": 0.18157163262367249, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18964433670043945, "sampling/sampling_logp_difference/max": 1.1991996765136719, "sampling/importance_sampling_ratio/min": 0.30143535137176514, "sampling/importance_sampling_ratio/mean": 1.0206162929534912, "sampling/importance_sampling_ratio/max": 1.9404109716415405, "entropy": 1.5031388700008392, "clip_ratio/low_mean": 0.08501362334936857, "clip_ratio/low_min": 0.08501362334936857, "clip_ratio/high_mean": 0.06383304297924042, "clip_ratio/high_max": 0.06383304297924042, "clip_ratio/region_mean": 0.148846666328609, "reward_total_mean": 0.5732332468032837, "reward_meter_mean": 0.4323441982269287, "reward_meter_std": 0.4019242823123932, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9930334091186523, "reward_repeat_soft_std": 0.010946320369839668, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5732332468032837, "reward_total_composite_std": 0.18157166242599487} {"timestamp_utc": "2026-04-12T23:39:30Z", "mode": "train", "global_step": 600, "epoch": 0.06027122049221497, "loss": -0.122, "grad_norm": 14.70077896118164, "learning_rate": 8.184848484848486e-06, "num_tokens": 1117128.0, "completions/mean_length": 22.875, "completions/min_length": 21.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.875, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.3396894633769989, "rewards/meter/std": 0.27054813504219055, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9959490299224854, "rewards/repeat_soft/std": 0.003451649099588394, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5284551382064819, "rewards/total_composite/std": 0.12187274545431137, "reward": 0.5284551382064819, "reward_std": 0.12187275290489197, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08534523099660873, "sampling/sampling_logp_difference/max": 1.8806533813476562, "sampling/importance_sampling_ratio/min": 0.15249043703079224, "sampling/importance_sampling_ratio/mean": 1.0147019624710083, "sampling/importance_sampling_ratio/max": 1.6807019710540771, "entropy": 0.5174635015428066, "clip_ratio/low_mean": 0.0527597414329648, "clip_ratio/low_min": 0.0527597414329648, "clip_ratio/high_mean": 0.01515151560306549, "clip_ratio/high_max": 0.01515151560306549, "clip_ratio/region_mean": 0.06791125703603029, "reward_total_mean": 0.5284551382064819, "reward_meter_mean": 0.3396894633769989, "reward_meter_std": 0.27054813504219055, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9959490299224854, "reward_repeat_soft_std": 0.003451649099588394, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5284551382064819, "reward_total_composite_std": 0.12187274545431137} {"timestamp_utc": "2026-04-12T23:40:26Z", "mode": "eval", "global_step": 600, "epoch": 0.06027122049221497, "eval_loss": NaN, "eval_runtime": 55.906, "eval_samples_per_second": 1.431, "eval_steps_per_second": 0.179, "eval_num_tokens": 1117128.0, "eval_completions/mean_length": 69.775, "eval_completions/min_length": 29.4, "eval_completions/max_length": 185.1, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 58.212500762939456, "eval_completions/min_terminated_length": 29.4, "eval_completions/max_terminated_length": 100.7, "eval_rewards/meter/mean": 0.6024353981018067, "eval_rewards/meter/std": 0.3768821984529495, "eval_rewards/count_adherence/mean": 0.8897916495800018, "eval_rewards/count_adherence/std": 0.12769289687275887, "eval_rewards/hard_gate/mean": 0.9375, "eval_rewards/hard_gate/std": 0.1767766922712326, "eval_rewards/repeat_soft/mean": 0.9884024918079376, "eval_rewards/repeat_soft/std": 0.018460330041125416, "eval_rewards/judge_quality/mean": 0.4888750046491623, "eval_rewards/judge_quality/std": 0.21980297863483428, "eval_rewards/total_composite/mean": 0.6218005120754242, "eval_rewards/total_composite/std": 0.23753960579633712, "eval_reward": 0.6218005120754242, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.13253581523895264, "eval_sampling/sampling_logp_difference/max": 1.2300835609436036, "eval_sampling/importance_sampling_ratio/min": 0.2966489166021347, "eval_sampling/importance_sampling_ratio/mean": 1.043230700492859, "eval_sampling/importance_sampling_ratio/max": 1.5889584302902222, "eval_entropy": 1.8897279977798462, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6218005120754242, "eval_reward_meter_mean": 0.6024353981018067, "eval_reward_meter_std": 0.3768821984529495, "eval_reward_count_adherence_mean": 0.8897916495800018, "eval_reward_count_adherence_std": 0.12769289687275887, "eval_reward_hard_gate_mean": 0.9375, "eval_reward_hard_gate_std": 0.1767766922712326, "eval_reward_repeat_soft_mean": 0.9884024918079376, "eval_reward_repeat_soft_std": 0.018460330041125416, "eval_reward_judge_quality_mean": 0.4888750046491623, "eval_reward_judge_quality_std": 0.21980297863483428, "eval_reward_total_composite_mean": 0.6218005120754242, "eval_reward_total_composite_std": 0.23753960579633712} {"timestamp_utc": "2026-04-12T23:40:35Z", "mode": "train", "global_step": 601, "epoch": 0.06037167252636866, "loss": 0.088, "grad_norm": 12.274469375610352, "learning_rate": 8.181818181818183e-06, "num_tokens": 1119520.0, "completions/mean_length": 66.0, "completions/min_length": 58.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9658753275871277, "rewards/meter/std": 0.03835213929414749, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9978299140930176, "rewards/repeat_soft/std": 0.002695566276088357, "rewards/judge_quality/mean": 0.3087499737739563, "rewards/judge_quality/std": 0.1141975075006485, "rewards/total_composite/mean": 0.7395519018173218, "rewards/total_composite/std": 0.03383920341730118, "reward": 0.7395519018173218, "reward_std": 0.03383919224143028, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20297037065029144, "sampling/sampling_logp_difference/max": 1.2722949981689453, "sampling/importance_sampling_ratio/min": 0.28018784523010254, "sampling/importance_sampling_ratio/mean": 1.0378531217575073, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5688178837299347, "clip_ratio/low_mean": 0.09929301962256432, "clip_ratio/low_min": 0.09929301962256432, "clip_ratio/high_mean": 0.05368338618427515, "clip_ratio/high_max": 0.05368338618427515, "clip_ratio/region_mean": 0.15297640580683947, "reward_total_mean": 0.7395519018173218, "reward_meter_mean": 0.9658753275871277, "reward_meter_std": 0.03835213929414749, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9978299140930176, "reward_repeat_soft_std": 0.002695566276088357, "reward_judge_quality_mean": 0.3087499737739563, "reward_judge_quality_std": 0.1141975075006485, "reward_total_composite_mean": 0.7395519018173218, "reward_total_composite_std": 0.03383920341730118} {"timestamp_utc": "2026-04-12T23:40:43Z", "mode": "train", "global_step": 602, "epoch": 0.06047212456052235, "loss": -0.0163, "grad_norm": 8.796191215515137, "learning_rate": 8.17878787878788e-06, "num_tokens": 1122387.0, "completions/mean_length": 119.375, "completions/min_length": 103.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.375, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.34147247672080994, "rewards/meter/std": 0.2758810818195343, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.07715168595314026, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9880790114402771, "rewards/repeat_soft/std": 0.014653024263679981, "rewards/judge_quality/mean": 0.2887499928474426, "rewards/judge_quality/std": 0.11630470305681229, "rewards/total_composite/mean": 0.4097350835800171, "rewards/total_composite/std": 0.19394925236701965, "reward": 0.4097350835800171, "reward_std": 0.19394923746585846, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19267664849758148, "sampling/sampling_logp_difference/max": 1.6625280380249023, "sampling/importance_sampling_ratio/min": 0.21064680814743042, "sampling/importance_sampling_ratio/mean": 1.0458078384399414, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3374589681625366, "clip_ratio/low_mean": 0.03551354072988033, "clip_ratio/low_min": 0.03551354072988033, "clip_ratio/high_mean": 0.12000076286494732, "clip_ratio/high_max": 0.12000076286494732, "clip_ratio/region_mean": 0.15551430359482765, "reward_total_mean": 0.4097350835800171, "reward_meter_mean": 0.34147247672080994, "reward_meter_std": 0.2758810818195343, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.07715168595314026, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9880790114402771, "reward_repeat_soft_std": 0.014653024263679981, "reward_judge_quality_mean": 0.2887499928474426, "reward_judge_quality_std": 0.11630470305681229, "reward_total_composite_mean": 0.4097350835800171, "reward_total_composite_std": 0.19394925236701965} {"timestamp_utc": "2026-04-12T23:40:55Z", "mode": "train", "global_step": 603, "epoch": 0.06057257659467604, "loss": -0.191, "grad_norm": 3.2724859714508057, "learning_rate": 8.175757575757577e-06, "num_tokens": 1125009.0, "completions/mean_length": 185.75, "completions/min_length": 125.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 139.1428680419922, "completions/min_terminated_length": 125.0, "completions/max_terminated_length": 151.0, "rewards/meter/mean": 0.7999477386474609, "rewards/meter/std": 0.28700780868530273, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.08908707648515701, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9963942170143127, "rewards/repeat_soft/std": 0.0029565212316811085, "rewards/judge_quality/mean": 0.3474999964237213, "rewards/judge_quality/std": 0.19695903360843658, "rewards/total_composite/mean": 0.6138966083526611, "rewards/total_composite/std": 0.2837551534175873, "reward": 0.6138966083526611, "reward_std": 0.2837551236152649, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19276374578475952, "sampling/sampling_logp_difference/max": 1.2265644073486328, "sampling/importance_sampling_ratio/min": 0.29329851269721985, "sampling/importance_sampling_ratio/mean": 1.042125940322876, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3439479023218155, "clip_ratio/low_mean": 0.03632946312427521, "clip_ratio/low_min": 0.03632946312427521, "clip_ratio/high_mean": 0.1054144911468029, "clip_ratio/high_max": 0.1054144911468029, "clip_ratio/region_mean": 0.1417439542710781, "reward_total_mean": 0.6138966083526611, "reward_meter_mean": 0.7999477386474609, "reward_meter_std": 0.28700780868530273, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.08908707648515701, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9963942170143127, "reward_repeat_soft_std": 0.0029565212316811085, "reward_judge_quality_mean": 0.3474999964237213, "reward_judge_quality_std": 0.19695903360843658, "reward_total_composite_mean": 0.6138966083526611, "reward_total_composite_std": 0.2837551534175873} {"timestamp_utc": "2026-04-12T23:41:01Z", "mode": "train", "global_step": 604, "epoch": 0.060673028628829735, "loss": 0.0493, "grad_norm": 20.110218048095703, "learning_rate": 8.172727272727273e-06, "num_tokens": 1126577.0, "completions/mean_length": 41.0, "completions/min_length": 35.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.9341446161270142, "rewards/meter/std": 0.08335070312023163, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9930774569511414, "rewards/repeat_soft/std": 0.011435138061642647, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7967978715896606, "rewards/total_composite/std": 0.03811796382069588, "reward": 0.7967978715896606, "reward_std": 0.03811794891953468, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16559156775474548, "sampling/sampling_logp_difference/max": 1.690549373626709, "sampling/importance_sampling_ratio/min": 0.2295154184103012, "sampling/importance_sampling_ratio/mean": 1.024551510810852, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3298236653208733, "clip_ratio/low_mean": 0.04029304161667824, "clip_ratio/low_min": 0.04029304161667824, "clip_ratio/high_mean": 0.09463870991021395, "clip_ratio/high_max": 0.09463870991021395, "clip_ratio/region_mean": 0.13493175152689219, "reward_total_mean": 0.7967978715896606, "reward_meter_mean": 0.9341446161270142, "reward_meter_std": 0.08335070312023163, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9930774569511414, "reward_repeat_soft_std": 0.011435138061642647, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7967978715896606, "reward_total_composite_std": 0.03811796382069588} {"timestamp_utc": "2026-04-12T23:41:08Z", "mode": "train", "global_step": 605, "epoch": 0.06077348066298342, "loss": 0.0152, "grad_norm": 14.3847017288208, "learning_rate": 8.16969696969697e-06, "num_tokens": 1128191.0, "completions/mean_length": 40.75, "completions/min_length": 37.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.75, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.8936046361923218, "rewards/meter/std": 0.18043449521064758, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9925197958946228, "rewards/repeat_soft/std": 0.0084422267973423, "rewards/judge_quality/mean": 0.48124998807907104, "rewards/judge_quality/std": 0.1606627106666565, "rewards/total_composite/mean": 0.7957490086555481, "rewards/total_composite/std": 0.06360577046871185, "reward": 0.7957490086555481, "reward_std": 0.06360577046871185, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20841501653194427, "sampling/sampling_logp_difference/max": 1.690964698791504, "sampling/importance_sampling_ratio/min": 0.18434159457683563, "sampling/importance_sampling_ratio/mean": 1.0438799858093262, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0925982743501663, "clip_ratio/low_mean": 0.05782632064074278, "clip_ratio/low_min": 0.05782632064074278, "clip_ratio/high_mean": 0.10941528342664242, "clip_ratio/high_max": 0.10941528342664242, "clip_ratio/region_mean": 0.1672416040673852, "reward_total_mean": 0.7957490086555481, "reward_meter_mean": 0.8936046361923218, "reward_meter_std": 0.18043449521064758, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9925197958946228, "reward_repeat_soft_std": 0.0084422267973423, "reward_judge_quality_mean": 0.48124998807907104, "reward_judge_quality_std": 0.1606627106666565, "reward_total_composite_mean": 0.7957490086555481, "reward_total_composite_std": 0.06360577046871185} {"timestamp_utc": "2026-04-12T23:41:14Z", "mode": "train", "global_step": 606, "epoch": 0.06087393269713712, "loss": 0.0113, "grad_norm": 22.281841278076172, "learning_rate": 8.166666666666668e-06, "num_tokens": 1129649.0, "completions/mean_length": 32.25, "completions/min_length": 27.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.25, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.6073384284973145, "rewards/meter/std": 0.33202311396598816, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9955383539199829, "rewards/repeat_soft/std": 0.006889368873089552, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.6976061463356018, "rewards/total_composite/std": 0.1599050611257553, "reward": 0.6976061463356018, "reward_std": 0.1599050611257553, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2040325403213501, "sampling/sampling_logp_difference/max": 2.4969639778137207, "sampling/importance_sampling_ratio/min": 0.08233458548784256, "sampling/importance_sampling_ratio/mean": 1.024932861328125, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5374352186918259, "clip_ratio/low_mean": 0.12480607675388455, "clip_ratio/low_min": 0.12480607675388455, "clip_ratio/high_mean": 0.0631472347304225, "clip_ratio/high_max": 0.0631472347304225, "clip_ratio/region_mean": 0.18795331148430705, "reward_total_mean": 0.6976061463356018, "reward_meter_mean": 0.6073384284973145, "reward_meter_std": 0.33202311396598816, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9955383539199829, "reward_repeat_soft_std": 0.006889368873089552, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.6976061463356018, "reward_total_composite_std": 0.1599050611257553} {"timestamp_utc": "2026-04-12T23:41:24Z", "mode": "train", "global_step": 607, "epoch": 0.060974384731290805, "loss": -0.0335, "grad_norm": 16.515682220458984, "learning_rate": 8.163636363636365e-06, "num_tokens": 1131384.0, "completions/mean_length": 57.875, "completions/min_length": 42.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.875, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.6567770838737488, "rewards/meter/std": 0.30241158604621887, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9972179532051086, "rewards/repeat_soft/std": 0.002561326837167144, "rewards/judge_quality/mean": 0.6112499833106995, "rewards/judge_quality/std": 0.15869447588920593, "rewards/total_composite/mean": 0.7239590287208557, "rewards/total_composite/std": 0.16300776600837708, "reward": 0.7239590287208557, "reward_std": 0.16300775110721588, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12924128770828247, "sampling/sampling_logp_difference/max": 1.5210199356079102, "sampling/importance_sampling_ratio/min": 0.21848894655704498, "sampling/importance_sampling_ratio/mean": 1.0113024711608887, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8074200376868248, "clip_ratio/low_mean": 0.046239178627729416, "clip_ratio/low_min": 0.046239178627729416, "clip_ratio/high_mean": 0.08635239582508802, "clip_ratio/high_max": 0.08635239582508802, "clip_ratio/region_mean": 0.13259157445281744, "reward_total_mean": 0.7239590287208557, "reward_meter_mean": 0.6567770838737488, "reward_meter_std": 0.30241158604621887, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9972179532051086, "reward_repeat_soft_std": 0.002561326837167144, "reward_judge_quality_mean": 0.6112499833106995, "reward_judge_quality_std": 0.15869447588920593, "reward_total_composite_mean": 0.7239590287208557, "reward_total_composite_std": 0.16300776600837708} {"timestamp_utc": "2026-04-12T23:41:30Z", "mode": "train", "global_step": 608, "epoch": 0.0610748367654445, "loss": -0.0199, "grad_norm": 15.279250144958496, "learning_rate": 8.16060606060606e-06, "num_tokens": 1132840.0, "completions/mean_length": 35.0, "completions/min_length": 30.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.921764612197876, "rewards/meter/std": 0.1367364376783371, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9870864152908325, "rewards/repeat_soft/std": 0.019315963611006737, "rewards/judge_quality/mean": 0.5074999928474426, "rewards/judge_quality/std": 0.18077215552330017, "rewards/total_composite/mean": 0.8157527446746826, "rewards/total_composite/std": 0.08696477860212326, "reward": 0.8157527446746826, "reward_std": 0.08696477115154266, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16159076988697052, "sampling/sampling_logp_difference/max": 1.3266401290893555, "sampling/importance_sampling_ratio/min": 0.2653673589229584, "sampling/importance_sampling_ratio/mean": 0.9908310174942017, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.108655720949173, "clip_ratio/low_mean": 0.09972210135310888, "clip_ratio/low_min": 0.09972210135310888, "clip_ratio/high_mean": 0.027799228206276894, "clip_ratio/high_max": 0.027799228206276894, "clip_ratio/region_mean": 0.12752132955938578, "reward_total_mean": 0.8157527446746826, "reward_meter_mean": 0.921764612197876, "reward_meter_std": 0.1367364376783371, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9870864152908325, "reward_repeat_soft_std": 0.019315963611006737, "reward_judge_quality_mean": 0.5074999928474426, "reward_judge_quality_std": 0.18077215552330017, "reward_total_composite_mean": 0.8157527446746826, "reward_total_composite_std": 0.08696477860212326} {"timestamp_utc": "2026-04-12T23:41:36Z", "mode": "train", "global_step": 609, "epoch": 0.061175288799598194, "loss": 0.125, "grad_norm": 14.252791404724121, "learning_rate": 8.15757575757576e-06, "num_tokens": 1134448.0, "completions/mean_length": 45.0, "completions/min_length": 37.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.7313787937164307, "rewards/meter/std": 0.36073043942451477, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 1.0, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.21084439754486084, "rewards/total_composite/mean": 0.7309954762458801, "rewards/total_composite/std": 0.15642228722572327, "reward": 0.7309954762458801, "reward_std": 0.15642230212688446, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20475496351718903, "sampling/sampling_logp_difference/max": 1.3468828201293945, "sampling/importance_sampling_ratio/min": 0.26004961133003235, "sampling/importance_sampling_ratio/mean": 1.0502336025238037, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.184522107243538, "clip_ratio/low_mean": 0.08310355991125107, "clip_ratio/low_min": 0.08310355991125107, "clip_ratio/high_mean": 0.08319156244397163, "clip_ratio/high_max": 0.08319156244397163, "clip_ratio/region_mean": 0.1662951223552227, "reward_total_mean": 0.7309954762458801, "reward_meter_mean": 0.7313787937164307, "reward_meter_std": 0.36073043942451477, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 1.0, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.21084439754486084, "reward_total_composite_mean": 0.7309954762458801, "reward_total_composite_std": 0.15642228722572327} {"timestamp_utc": "2026-04-12T23:41:42Z", "mode": "train", "global_step": 610, "epoch": 0.06127574083375188, "loss": 0.0499, "grad_norm": 18.17657470703125, "learning_rate": 8.154545454545455e-06, "num_tokens": 1135842.0, "completions/mean_length": 28.25, "completions/min_length": 22.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.25, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.683879017829895, "rewards/meter/std": 0.38403958082199097, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.995398998260498, "rewards/repeat_soft/std": 0.009016699157655239, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.5996274948120117, "rewards/total_composite/std": 0.30776020884513855, "reward": 0.5996274948120117, "reward_std": 0.30776020884513855, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20415687561035156, "sampling/sampling_logp_difference/max": 1.542629361152649, "sampling/importance_sampling_ratio/min": 0.21381816267967224, "sampling/importance_sampling_ratio/mean": 1.042561650276184, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5655753761529922, "clip_ratio/low_mean": 0.09382183849811554, "clip_ratio/low_min": 0.09382183849811554, "clip_ratio/high_mean": 0.09351345337927341, "clip_ratio/high_max": 0.09351345337927341, "clip_ratio/region_mean": 0.18733529187738895, "reward_total_mean": 0.5996274948120117, "reward_meter_mean": 0.683879017829895, "reward_meter_std": 0.38403958082199097, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.995398998260498, "reward_repeat_soft_std": 0.009016699157655239, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.5996274948120117, "reward_total_composite_std": 0.30776020884513855} {"timestamp_utc": "2026-04-12T23:41:48Z", "mode": "train", "global_step": 611, "epoch": 0.06137619286790558, "loss": 0.007, "grad_norm": 16.28204345703125, "learning_rate": 8.151515151515152e-06, "num_tokens": 1137299.0, "completions/mean_length": 37.125, "completions/min_length": 32.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.125, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.7204763889312744, "rewards/meter/std": 0.41953638195991516, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.993358850479126, "rewards/repeat_soft/std": 0.012442831881344318, "rewards/judge_quality/mean": 0.7487499713897705, "rewards/judge_quality/std": 0.21853001415729523, "rewards/total_composite/mean": 0.7981752157211304, "rewards/total_composite/std": 0.21186815202236176, "reward": 0.7981752157211304, "reward_std": 0.21186815202236176, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2027798444032669, "sampling/sampling_logp_difference/max": 1.9750261306762695, "sampling/importance_sampling_ratio/min": 0.13875769078731537, "sampling/importance_sampling_ratio/mean": 0.9910250306129456, "sampling/importance_sampling_ratio/max": 1.9572968482971191, "entropy": 1.5063078626990318, "clip_ratio/low_mean": 0.04050164483487606, "clip_ratio/low_min": 0.04050164483487606, "clip_ratio/high_mean": 0.1567803304642439, "clip_ratio/high_max": 0.1567803304642439, "clip_ratio/region_mean": 0.19728197529911995, "reward_total_mean": 0.7981752157211304, "reward_meter_mean": 0.7204763889312744, "reward_meter_std": 0.41953638195991516, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.993358850479126, "reward_repeat_soft_std": 0.012442831881344318, "reward_judge_quality_mean": 0.7487499713897705, "reward_judge_quality_std": 0.21853001415729523, "reward_total_composite_mean": 0.7981752157211304, "reward_total_composite_std": 0.21186815202236176} {"timestamp_utc": "2026-04-12T23:41:55Z", "mode": "train", "global_step": 612, "epoch": 0.061476644902059265, "loss": 0.0089, "grad_norm": 13.099044799804688, "learning_rate": 8.14848484848485e-06, "num_tokens": 1139040.0, "completions/mean_length": 60.625, "completions/min_length": 50.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.625, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.5470455884933472, "rewards/meter/std": 0.37272265553474426, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9909868240356445, "rewards/repeat_soft/std": 0.01370298396795988, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.6400191783905029, "rewards/total_composite/std": 0.1628931760787964, "reward": 0.6400191783905029, "reward_std": 0.1628931760787964, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19092382490634918, "sampling/sampling_logp_difference/max": 2.0902814865112305, "sampling/importance_sampling_ratio/min": 0.12365231662988663, "sampling/importance_sampling_ratio/mean": 1.0298277139663696, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0493016242980957, "clip_ratio/low_mean": 0.0896091777831316, "clip_ratio/low_min": 0.0896091777831316, "clip_ratio/high_mean": 0.0882062055170536, "clip_ratio/high_max": 0.0882062055170536, "clip_ratio/region_mean": 0.1778153833001852, "reward_total_mean": 0.6400191783905029, "reward_meter_mean": 0.5470455884933472, "reward_meter_std": 0.37272265553474426, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9909868240356445, "reward_repeat_soft_std": 0.01370298396795988, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.6400191783905029, "reward_total_composite_std": 0.1628931760787964} {"timestamp_utc": "2026-04-12T23:42:01Z", "mode": "train", "global_step": 613, "epoch": 0.06157709693621296, "loss": -0.0086, "grad_norm": 19.023317337036133, "learning_rate": 8.145454545454547e-06, "num_tokens": 1140452.0, "completions/mean_length": 19.5, "completions/min_length": 16.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.5, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.27833446860313416, "rewards/meter/std": 0.43387213349342346, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9594928026199341, "rewards/repeat_soft/std": 0.008505655452609062, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.0975411981344223, "rewards/total_composite/mean": 0.48819980025291443, "rewards/total_composite/std": 0.20400124788284302, "reward": 0.48819980025291443, "reward_std": 0.20400121808052063, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18322478234767914, "sampling/sampling_logp_difference/max": 1.5562248229980469, "sampling/importance_sampling_ratio/min": 0.21093086898326874, "sampling/importance_sampling_ratio/mean": 1.0519344806671143, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.402065135538578, "clip_ratio/low_mean": 0.11251747980713844, "clip_ratio/low_min": 0.11251747980713844, "clip_ratio/high_mean": 0.0357142873108387, "clip_ratio/high_max": 0.0357142873108387, "clip_ratio/region_mean": 0.14823176711797714, "reward_total_mean": 0.48819980025291443, "reward_meter_mean": 0.27833446860313416, "reward_meter_std": 0.43387213349342346, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9594928026199341, "reward_repeat_soft_std": 0.008505655452609062, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.0975411981344223, "reward_total_composite_mean": 0.48819980025291443, "reward_total_composite_std": 0.20400124788284302} {"timestamp_utc": "2026-04-12T23:42:07Z", "mode": "train", "global_step": 614, "epoch": 0.06167754897036665, "loss": 0.1499, "grad_norm": 17.97779655456543, "learning_rate": 8.142424242424242e-06, "num_tokens": 1141932.0, "completions/mean_length": 38.0, "completions/min_length": 29.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.8238444328308105, "rewards/meter/std": 0.3374292254447937, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9962548017501831, "rewards/repeat_soft/std": 0.00590617535635829, "rewards/judge_quality/mean": 0.5612499713897705, "rewards/judge_quality/std": 0.2032899260520935, "rewards/total_composite/mean": 0.7887305021286011, "rewards/total_composite/std": 0.18120287358760834, "reward": 0.7887305021286011, "reward_std": 0.18120285868644714, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18676792085170746, "sampling/sampling_logp_difference/max": 1.255594253540039, "sampling/importance_sampling_ratio/min": 0.2849065065383911, "sampling/importance_sampling_ratio/mean": 1.0470454692840576, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7776105850934982, "clip_ratio/low_mean": 0.048124998807907104, "clip_ratio/low_min": 0.048124998807907104, "clip_ratio/high_mean": 0.10777550097554922, "clip_ratio/high_max": 0.10777550097554922, "clip_ratio/region_mean": 0.15590049978345633, "reward_total_mean": 0.7887305021286011, "reward_meter_mean": 0.8238444328308105, "reward_meter_std": 0.3374292254447937, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9962548017501831, "reward_repeat_soft_std": 0.00590617535635829, "reward_judge_quality_mean": 0.5612499713897705, "reward_judge_quality_std": 0.2032899260520935, "reward_total_composite_mean": 0.7887305021286011, "reward_total_composite_std": 0.18120287358760834} {"timestamp_utc": "2026-04-12T23:42:14Z", "mode": "train", "global_step": 615, "epoch": 0.06177800100452034, "loss": -0.0831, "grad_norm": 9.735858917236328, "learning_rate": 8.139393939393941e-06, "num_tokens": 1144046.0, "completions/mean_length": 72.25, "completions/min_length": 53.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.25, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.9169949293136597, "rewards/meter/std": 0.15882092714309692, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9103500843048096, "rewards/repeat_soft/std": 0.08497316390275955, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.6368836164474487, "rewards/total_composite/std": 0.2692444324493408, "reward": 0.6368836164474487, "reward_std": 0.26924440264701843, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1715807020664215, "sampling/sampling_logp_difference/max": 1.4919424057006836, "sampling/importance_sampling_ratio/min": 0.22493532299995422, "sampling/importance_sampling_ratio/mean": 1.0403074026107788, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8049511760473251, "clip_ratio/low_mean": 0.05404732283204794, "clip_ratio/low_min": 0.05404732283204794, "clip_ratio/high_mean": 0.13147368654608727, "clip_ratio/high_max": 0.13147368654608727, "clip_ratio/region_mean": 0.1855210093781352, "reward_total_mean": 0.6368836164474487, "reward_meter_mean": 0.9169949293136597, "reward_meter_std": 0.15882092714309692, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9103500843048096, "reward_repeat_soft_std": 0.08497316390275955, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.6368836164474487, "reward_total_composite_std": 0.2692444324493408} {"timestamp_utc": "2026-04-12T23:42:21Z", "mode": "train", "global_step": 616, "epoch": 0.061878453038674036, "loss": -0.0343, "grad_norm": 16.79818344116211, "learning_rate": 8.136363636363637e-06, "num_tokens": 1145511.0, "completions/mean_length": 37.125, "completions/min_length": 31.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.125, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.7454337477684021, "rewards/meter/std": 0.3255024552345276, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9977384209632874, "rewards/repeat_soft/std": 0.004657295998185873, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7123440504074097, "rewards/total_composite/std": 0.14733818173408508, "reward": 0.7123440504074097, "reward_std": 0.14733819663524628, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18304798007011414, "sampling/sampling_logp_difference/max": 1.1938247680664062, "sampling/importance_sampling_ratio/min": 0.3030599057674408, "sampling/importance_sampling_ratio/mean": 1.027708888053894, "sampling/importance_sampling_ratio/max": 1.9923049211502075, "entropy": 1.549539640545845, "clip_ratio/low_mean": 0.060483869165182114, "clip_ratio/low_min": 0.060483869165182114, "clip_ratio/high_mean": 0.1489045824855566, "clip_ratio/high_max": 0.1489045824855566, "clip_ratio/region_mean": 0.20938845165073872, "reward_total_mean": 0.7123440504074097, "reward_meter_mean": 0.7454337477684021, "reward_meter_std": 0.3255024552345276, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9977384209632874, "reward_repeat_soft_std": 0.004657295998185873, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7123440504074097, "reward_total_composite_std": 0.14733818173408508} {"timestamp_utc": "2026-04-12T23:42:27Z", "mode": "train", "global_step": 617, "epoch": 0.061978905072827724, "loss": 0.0811, "grad_norm": 18.3663272857666, "learning_rate": 8.133333333333334e-06, "num_tokens": 1146926.0, "completions/mean_length": 28.875, "completions/min_length": 24.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.875, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.594732403755188, "rewards/meter/std": 0.35402271151542664, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9952442646026611, "rewards/repeat_soft/std": 0.006869346369057894, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.6412789821624756, "rewards/total_composite/std": 0.15332868695259094, "reward": 0.6412789821624756, "reward_std": 0.15332865715026855, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1670580506324768, "sampling/sampling_logp_difference/max": 1.6013193130493164, "sampling/importance_sampling_ratio/min": 0.231222465634346, "sampling/importance_sampling_ratio/mean": 0.9977790713310242, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.144897200167179, "clip_ratio/low_mean": 0.037124582566320896, "clip_ratio/low_min": 0.037124582566320896, "clip_ratio/high_mean": 0.06636051367968321, "clip_ratio/high_max": 0.06636051367968321, "clip_ratio/region_mean": 0.1034850962460041, "reward_total_mean": 0.6412789821624756, "reward_meter_mean": 0.594732403755188, "reward_meter_std": 0.35402271151542664, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9952442646026611, "reward_repeat_soft_std": 0.006869346369057894, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.6412789821624756, "reward_total_composite_std": 0.15332868695259094} {"timestamp_utc": "2026-04-12T23:42:34Z", "mode": "train", "global_step": 618, "epoch": 0.06207935710698142, "loss": 0.1092, "grad_norm": 13.461973190307617, "learning_rate": 8.130303030303031e-06, "num_tokens": 1148698.0, "completions/mean_length": 58.5, "completions/min_length": 46.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.5, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.787352442741394, "rewards/meter/std": 0.22498290240764618, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9853062033653259, "rewards/repeat_soft/std": 0.010363499633967876, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.5576370358467102, "rewards/total_composite/std": 0.3514828383922577, "reward": 0.5576370358467102, "reward_std": 0.3514828085899353, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21458286046981812, "sampling/sampling_logp_difference/max": 1.7027578353881836, "sampling/importance_sampling_ratio/min": 0.18218040466308594, "sampling/importance_sampling_ratio/mean": 1.0436837673187256, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.416036009788513, "clip_ratio/low_mean": 0.030679156072437763, "clip_ratio/low_min": 0.030679156072437763, "clip_ratio/high_mean": 0.12249129265546799, "clip_ratio/high_max": 0.12249129265546799, "clip_ratio/region_mean": 0.15317044872790575, "reward_total_mean": 0.5576370358467102, "reward_meter_mean": 0.787352442741394, "reward_meter_std": 0.22498290240764618, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9853062033653259, "reward_repeat_soft_std": 0.010363499633967876, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.5576370358467102, "reward_total_composite_std": 0.3514828383922577} {"timestamp_utc": "2026-04-12T23:42:40Z", "mode": "train", "global_step": 619, "epoch": 0.062179809141135106, "loss": 0.0925, "grad_norm": 17.825748443603516, "learning_rate": 8.127272727272728e-06, "num_tokens": 1150109.0, "completions/mean_length": 29.375, "completions/min_length": 27.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.375, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.3541598916053772, "rewards/meter/std": 0.3066812753677368, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9913157820701599, "rewards/repeat_soft/std": 0.020058302208781242, "rewards/judge_quality/mean": 0.8575000166893005, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.6657534837722778, "rewards/total_composite/std": 0.09836650639772415, "reward": 0.6657534837722778, "reward_std": 0.09836649894714355, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12300041317939758, "sampling/sampling_logp_difference/max": 1.580701470375061, "sampling/importance_sampling_ratio/min": 0.2058306634426117, "sampling/importance_sampling_ratio/mean": 0.9927932024002075, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6258425749838352, "clip_ratio/low_mean": 0.059378146193921566, "clip_ratio/low_min": 0.059378146193921566, "clip_ratio/high_mean": 0.06417624466121197, "clip_ratio/high_max": 0.06417624466121197, "clip_ratio/region_mean": 0.12355439085513353, "reward_total_mean": 0.6657534837722778, "reward_meter_mean": 0.3541598916053772, "reward_meter_std": 0.3066812753677368, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9913157820701599, "reward_repeat_soft_std": 0.020058302208781242, "reward_judge_quality_mean": 0.8575000166893005, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.6657534837722778, "reward_total_composite_std": 0.09836650639772415} {"timestamp_utc": "2026-04-12T23:42:47Z", "mode": "train", "global_step": 620, "epoch": 0.0622802611752888, "loss": 0.0079, "grad_norm": 11.035258293151855, "learning_rate": 8.124242424242424e-06, "num_tokens": 1152203.0, "completions/mean_length": 73.75, "completions/min_length": 48.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.75, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.5664907097816467, "rewards/meter/std": 0.33263924717903137, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9825822114944458, "rewards/repeat_soft/std": 0.01900266855955124, "rewards/judge_quality/mean": 0.5074999928474426, "rewards/judge_quality/std": 0.1642080694437027, "rewards/total_composite/mean": 0.6507415771484375, "rewards/total_composite/std": 0.14177948236465454, "reward": 0.6507415771484375, "reward_std": 0.14177949726581573, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17959293723106384, "sampling/sampling_logp_difference/max": 1.257352352142334, "sampling/importance_sampling_ratio/min": 0.28440603613853455, "sampling/importance_sampling_ratio/mean": 1.04653799533844, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2396499812602997, "clip_ratio/low_mean": 0.06999729387462139, "clip_ratio/low_min": 0.06999729387462139, "clip_ratio/high_mean": 0.09751482494175434, "clip_ratio/high_max": 0.09751482494175434, "clip_ratio/region_mean": 0.16751211881637573, "reward_total_mean": 0.6507415771484375, "reward_meter_mean": 0.5664907097816467, "reward_meter_std": 0.33263924717903137, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9825822114944458, "reward_repeat_soft_std": 0.01900266855955124, "reward_judge_quality_mean": 0.5074999928474426, "reward_judge_quality_std": 0.1642080694437027, "reward_total_composite_mean": 0.6507415771484375, "reward_total_composite_std": 0.14177948236465454} {"timestamp_utc": "2026-04-12T23:42:53Z", "mode": "train", "global_step": 621, "epoch": 0.06238071320944249, "loss": 0.0054, "grad_norm": 15.580313682556152, "learning_rate": 8.121212121212121e-06, "num_tokens": 1153920.0, "completions/mean_length": 45.625, "completions/min_length": 35.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.625, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.7237141132354736, "rewards/meter/std": 0.3663863241672516, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9993371367454529, "rewards/repeat_soft/std": 0.0018749026348814368, "rewards/judge_quality/mean": 0.4737499952316284, "rewards/judge_quality/std": 0.16291432082653046, "rewards/total_composite/mean": 0.7177301049232483, "rewards/total_composite/std": 0.1600118726491928, "reward": 0.7177301049232483, "reward_std": 0.1600118726491928, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22073960304260254, "sampling/sampling_logp_difference/max": 1.9725987911224365, "sampling/importance_sampling_ratio/min": 0.13909490406513214, "sampling/importance_sampling_ratio/mean": 1.0361422300338745, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9244915693998337, "clip_ratio/low_mean": 0.06334325484931469, "clip_ratio/low_min": 0.06334325484931469, "clip_ratio/high_mean": 0.1550257559865713, "clip_ratio/high_max": 0.1550257559865713, "clip_ratio/region_mean": 0.218369010835886, "reward_total_mean": 0.7177301049232483, "reward_meter_mean": 0.7237141132354736, "reward_meter_std": 0.3663863241672516, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9993371367454529, "reward_repeat_soft_std": 0.0018749026348814368, "reward_judge_quality_mean": 0.4737499952316284, "reward_judge_quality_std": 0.16291432082653046, "reward_total_composite_mean": 0.7177301049232483, "reward_total_composite_std": 0.1600118726491928} {"timestamp_utc": "2026-04-12T23:43:04Z", "mode": "train", "global_step": 622, "epoch": 0.06248116524359618, "loss": -0.1591, "grad_norm": 3.234005928039551, "learning_rate": 8.118181818181819e-06, "num_tokens": 1155727.0, "completions/mean_length": 123.875, "completions/min_length": 57.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 68.42857360839844, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.6806365847587585, "rewards/meter/std": 0.39670050144195557, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9981823563575745, "rewards/repeat_soft/std": 0.002973742550238967, "rewards/judge_quality/mean": 0.3712500035762787, "rewards/judge_quality/std": 0.15037453174591064, "rewards/total_composite/mean": 0.6349983215332031, "rewards/total_composite/std": 0.2886177599430084, "reward": 0.6349983215332031, "reward_std": 0.2886177599430084, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17926889657974243, "sampling/sampling_logp_difference/max": 1.0793695449829102, "sampling/importance_sampling_ratio/min": 0.33980968594551086, "sampling/importance_sampling_ratio/mean": 1.049180269241333, "sampling/importance_sampling_ratio/max": 1.9916385412216187, "entropy": 1.5097311735153198, "clip_ratio/low_mean": 0.015625, "clip_ratio/low_min": 0.015625, "clip_ratio/high_mean": 0.10660857753828168, "clip_ratio/high_max": 0.10660857753828168, "clip_ratio/region_mean": 0.12223357753828168, "reward_total_mean": 0.6349983215332031, "reward_meter_mean": 0.6806365847587585, "reward_meter_std": 0.39670050144195557, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9981823563575745, "reward_repeat_soft_std": 0.002973742550238967, "reward_judge_quality_mean": 0.3712500035762787, "reward_judge_quality_std": 0.15037453174591064, "reward_total_composite_mean": 0.6349983215332031, "reward_total_composite_std": 0.2886177599430084} {"timestamp_utc": "2026-04-12T23:43:11Z", "mode": "train", "global_step": 623, "epoch": 0.06258161727774987, "loss": 0.0253, "grad_norm": 16.074350357055664, "learning_rate": 8.115151515151516e-06, "num_tokens": 1157423.0, "completions/mean_length": 38.0, "completions/min_length": 35.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9654180407524109, "rewards/meter/std": 0.03988801687955856, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9771788120269775, "rewards/repeat_soft/std": 0.03744228929281235, "rewards/judge_quality/mean": 0.38874998688697815, "rewards/judge_quality/std": 0.08675704896450043, "rewards/total_composite/mean": 0.7987809777259827, "rewards/total_composite/std": 0.03702932223677635, "reward": 0.7987809777259827, "reward_std": 0.03702932223677635, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1946445107460022, "sampling/sampling_logp_difference/max": 2.062849998474121, "sampling/importance_sampling_ratio/min": 0.12709124386310577, "sampling/importance_sampling_ratio/mean": 1.0245599746704102, "sampling/importance_sampling_ratio/max": 1.8117117881774902, "entropy": 1.4282769709825516, "clip_ratio/low_mean": 0.053418803960084915, "clip_ratio/low_min": 0.053418803960084915, "clip_ratio/high_mean": 0.13562512304633856, "clip_ratio/high_max": 0.13562512304633856, "clip_ratio/region_mean": 0.18904392700642347, "reward_total_mean": 0.7987809777259827, "reward_meter_mean": 0.9654180407524109, "reward_meter_std": 0.03988801687955856, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9771788120269775, "reward_repeat_soft_std": 0.03744228929281235, "reward_judge_quality_mean": 0.38874998688697815, "reward_judge_quality_std": 0.08675704896450043, "reward_total_composite_mean": 0.7987809777259827, "reward_total_composite_std": 0.03702932223677635} {"timestamp_utc": "2026-04-12T23:43:17Z", "mode": "train", "global_step": 624, "epoch": 0.06268206931190357, "loss": 0.1115, "grad_norm": 32.002830505371094, "learning_rate": 8.112121212121213e-06, "num_tokens": 1158735.0, "completions/mean_length": 17.0, "completions/min_length": 11.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 17.0, "completions/min_terminated_length": 11.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.5454722046852112, "rewards/meter/std": 0.375191867351532, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.375, "rewards/judge_quality/std": 0.09411238878965378, "rewards/total_composite/mean": 0.6042125225067139, "rewards/total_composite/std": 0.16663211584091187, "reward": 0.6042125225067139, "reward_std": 0.16663208603858948, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2352629005908966, "sampling/sampling_logp_difference/max": 1.8035848140716553, "sampling/importance_sampling_ratio/min": 0.16470739245414734, "sampling/importance_sampling_ratio/mean": 1.0321533679962158, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.068661689758301, "clip_ratio/low_mean": 0.12912333011627197, "clip_ratio/low_min": 0.12912333011627197, "clip_ratio/high_mean": 0.07500000204890966, "clip_ratio/high_max": 0.07500000204890966, "clip_ratio/region_mean": 0.20412333216518164, "reward_total_mean": 0.6042125225067139, "reward_meter_mean": 0.5454722046852112, "reward_meter_std": 0.375191867351532, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.375, "reward_judge_quality_std": 0.09411238878965378, "reward_total_composite_mean": 0.6042125225067139, "reward_total_composite_std": 0.16663211584091187} {"timestamp_utc": "2026-04-12T23:43:23Z", "mode": "train", "global_step": 625, "epoch": 0.06278252134605726, "loss": 0.1143, "grad_norm": 22.57779884338379, "learning_rate": 8.10909090909091e-06, "num_tokens": 1160240.0, "completions/mean_length": 35.125, "completions/min_length": 31.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.125, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.6660373210906982, "rewards/meter/std": 0.37584179639816284, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9917200803756714, "rewards/repeat_soft/std": 0.007631211541593075, "rewards/judge_quality/mean": 0.6274999976158142, "rewards/judge_quality/std": 0.2253093272447586, "rewards/total_composite/mean": 0.7371387481689453, "rewards/total_composite/std": 0.13217169046401978, "reward": 0.7371387481689453, "reward_std": 0.13217169046401978, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1646099090576172, "sampling/sampling_logp_difference/max": 1.5092792510986328, "sampling/importance_sampling_ratio/min": 0.22106926143169403, "sampling/importance_sampling_ratio/mean": 1.0396641492843628, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2431165054440498, "clip_ratio/low_mean": 0.04090624861419201, "clip_ratio/low_min": 0.04090624861419201, "clip_ratio/high_mean": 0.10690709389746189, "clip_ratio/high_max": 0.10690709389746189, "clip_ratio/region_mean": 0.1478133425116539, "reward_total_mean": 0.7371387481689453, "reward_meter_mean": 0.6660373210906982, "reward_meter_std": 0.37584179639816284, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9917200803756714, "reward_repeat_soft_std": 0.007631211541593075, "reward_judge_quality_mean": 0.6274999976158142, "reward_judge_quality_std": 0.2253093272447586, "reward_total_composite_mean": 0.7371387481689453, "reward_total_composite_std": 0.13217169046401978} {"timestamp_utc": "2026-04-12T23:43:30Z", "mode": "train", "global_step": 626, "epoch": 0.06288297338021095, "loss": 0.0345, "grad_norm": 20.278907775878906, "learning_rate": 8.106060606060606e-06, "num_tokens": 1161726.0, "completions/mean_length": 33.75, "completions/min_length": 31.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.75, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.8673355579376221, "rewards/meter/std": 0.18751801550388336, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9960410594940186, "rewards/repeat_soft/std": 0.006024917121976614, "rewards/judge_quality/mean": 0.5762499570846558, "rewards/judge_quality/std": 0.19827382266521454, "rewards/total_composite/mean": 0.8127800822257996, "rewards/total_composite/std": 0.09588909894227982, "reward": 0.8127800822257996, "reward_std": 0.09588909894227982, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19125910103321075, "sampling/sampling_logp_difference/max": 1.4346199035644531, "sampling/importance_sampling_ratio/min": 0.23820587992668152, "sampling/importance_sampling_ratio/mean": 1.0260168313980103, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7419205456972122, "clip_ratio/low_mean": 0.09806179301813245, "clip_ratio/low_min": 0.09806179301813245, "clip_ratio/high_mean": 0.05543592572212219, "clip_ratio/high_max": 0.05543592572212219, "clip_ratio/region_mean": 0.15349771874025464, "reward_total_mean": 0.8127800822257996, "reward_meter_mean": 0.8673355579376221, "reward_meter_std": 0.18751801550388336, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9960410594940186, "reward_repeat_soft_std": 0.006024917121976614, "reward_judge_quality_mean": 0.5762499570846558, "reward_judge_quality_std": 0.19827382266521454, "reward_total_composite_mean": 0.8127800822257996, "reward_total_composite_std": 0.09588909894227982} {"timestamp_utc": "2026-04-12T23:43:37Z", "mode": "train", "global_step": 627, "epoch": 0.06298342541436464, "loss": 0.0069, "grad_norm": 11.530714988708496, "learning_rate": 8.103030303030303e-06, "num_tokens": 1163817.0, "completions/mean_length": 91.375, "completions/min_length": 84.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.375, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.9446588754653931, "rewards/meter/std": 0.10983403027057648, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9861116409301758, "rewards/repeat_soft/std": 0.014554635621607304, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7886452078819275, "rewards/total_composite/std": 0.06147077679634094, "reward": 0.7886452078819275, "reward_std": 0.06147078052163124, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20375370979309082, "sampling/sampling_logp_difference/max": 1.7005796432495117, "sampling/importance_sampling_ratio/min": 0.18257765471935272, "sampling/importance_sampling_ratio/mean": 1.0450621843338013, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.253380075097084, "clip_ratio/low_mean": 0.0366924162954092, "clip_ratio/low_min": 0.0366924162954092, "clip_ratio/high_mean": 0.1238513421267271, "clip_ratio/high_max": 0.1238513421267271, "clip_ratio/region_mean": 0.1605437584221363, "reward_total_mean": 0.7886452078819275, "reward_meter_mean": 0.9446588754653931, "reward_meter_std": 0.10983403027057648, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9861116409301758, "reward_repeat_soft_std": 0.014554635621607304, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7886452078819275, "reward_total_composite_std": 0.06147077679634094} {"timestamp_utc": "2026-04-12T23:43:49Z", "mode": "train", "global_step": 628, "epoch": 0.06308387744851833, "loss": -0.2122, "grad_norm": 2.1132583618164062, "learning_rate": 8.1e-06, "num_tokens": 1166151.0, "completions/mean_length": 173.75, "completions/min_length": 108.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 125.42857360839844, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 151.0, "rewards/meter/mean": 0.9160116910934448, "rewards/meter/std": 0.13900001347064972, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9517726898193359, "rewards/repeat_soft/std": 0.07515332847833633, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.13715866208076477, "rewards/total_composite/mean": 0.6910140514373779, "rewards/total_composite/std": 0.2815585136413574, "reward": 0.6910140514373779, "reward_std": 0.2815585136413574, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18184438347816467, "sampling/sampling_logp_difference/max": 1.5091943740844727, "sampling/importance_sampling_ratio/min": 0.24494464695453644, "sampling/importance_sampling_ratio/mean": 1.0318944454193115, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8570679426193237, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.1627912363037467, "clip_ratio/high_max": 0.1627912363037467, "clip_ratio/region_mean": 0.1627912363037467, "reward_total_mean": 0.6910140514373779, "reward_meter_mean": 0.9160116910934448, "reward_meter_std": 0.13900001347064972, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9517726898193359, "reward_repeat_soft_std": 0.07515332847833633, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.13715866208076477, "reward_total_composite_mean": 0.6910140514373779, "reward_total_composite_std": 0.2815585136413574} {"timestamp_utc": "2026-04-12T23:43:56Z", "mode": "train", "global_step": 629, "epoch": 0.06318432948267202, "loss": 0.0728, "grad_norm": 20.417993545532227, "learning_rate": 8.096969696969698e-06, "num_tokens": 1167666.0, "completions/mean_length": 36.375, "completions/min_length": 27.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.375, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.40054258704185486, "rewards/meter/std": 0.320362389087677, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9866778254508972, "rewards/repeat_soft/std": 0.023427648469805717, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.12906256318092346, "rewards/total_composite/mean": 0.5609119534492493, "rewards/total_composite/std": 0.14431042969226837, "reward": 0.5609119534492493, "reward_std": 0.14431042969226837, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20754337310791016, "sampling/sampling_logp_difference/max": 1.9730651378631592, "sampling/importance_sampling_ratio/min": 0.13903005421161652, "sampling/importance_sampling_ratio/mean": 1.0191223621368408, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.792641043663025, "clip_ratio/low_mean": 0.10948249138891697, "clip_ratio/low_min": 0.10948249138891697, "clip_ratio/high_mean": 0.053472223691642284, "clip_ratio/high_max": 0.053472223691642284, "clip_ratio/region_mean": 0.16295471508055925, "reward_total_mean": 0.5609119534492493, "reward_meter_mean": 0.40054258704185486, "reward_meter_std": 0.320362389087677, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9866778254508972, "reward_repeat_soft_std": 0.023427648469805717, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.12906256318092346, "reward_total_composite_mean": 0.5609119534492493, "reward_total_composite_std": 0.14431042969226837} {"timestamp_utc": "2026-04-12T23:44:03Z", "mode": "train", "global_step": 630, "epoch": 0.06328478151682572, "loss": 0.0284, "grad_norm": 9.638946533203125, "learning_rate": 8.093939393939395e-06, "num_tokens": 1169763.0, "completions/mean_length": 70.125, "completions/min_length": 61.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.125, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.6417567133903503, "rewards/meter/std": 0.32083654403686523, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9960492849349976, "rewards/repeat_soft/std": 0.004478702321648598, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.5935440063476562, "rewards/total_composite/std": 0.27082327008247375, "reward": 0.5935440063476562, "reward_std": 0.27082324028015137, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.176280677318573, "sampling/sampling_logp_difference/max": 1.2115707397460938, "sampling/importance_sampling_ratio/min": 0.2977292537689209, "sampling/importance_sampling_ratio/mean": 1.0173544883728027, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7082483172416687, "clip_ratio/low_mean": 0.0309404069557786, "clip_ratio/low_min": 0.0309404069557786, "clip_ratio/high_mean": 0.09007307887077332, "clip_ratio/high_max": 0.09007307887077332, "clip_ratio/region_mean": 0.12101348582655191, "reward_total_mean": 0.5935440063476562, "reward_meter_mean": 0.6417567133903503, "reward_meter_std": 0.32083654403686523, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9960492849349976, "reward_repeat_soft_std": 0.004478702321648598, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.5935440063476562, "reward_total_composite_std": 0.27082327008247375} {"timestamp_utc": "2026-04-12T23:44:10Z", "mode": "train", "global_step": 631, "epoch": 0.06338523355097941, "loss": 0.0631, "grad_norm": 15.66556453704834, "learning_rate": 8.090909090909092e-06, "num_tokens": 1171409.0, "completions/mean_length": 48.75, "completions/min_length": 43.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.75, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.26772063970565796, "rewards/meter/std": 0.30965161323547363, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9562331438064575, "rewards/repeat_soft/std": 0.042347077280282974, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.4857226014137268, "rewards/total_composite/std": 0.12838341295719147, "reward": 0.4857226014137268, "reward_std": 0.12838341295719147, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1586320847272873, "sampling/sampling_logp_difference/max": 1.8656830787658691, "sampling/importance_sampling_ratio/min": 0.15479044616222382, "sampling/importance_sampling_ratio/mean": 1.021510362625122, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9910558983683586, "clip_ratio/low_mean": 0.08387583773583174, "clip_ratio/low_min": 0.08387583773583174, "clip_ratio/high_mean": 0.058563945814967155, "clip_ratio/high_max": 0.058563945814967155, "clip_ratio/region_mean": 0.1424397835507989, "reward_total_mean": 0.4857226014137268, "reward_meter_mean": 0.26772063970565796, "reward_meter_std": 0.30965161323547363, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9562331438064575, "reward_repeat_soft_std": 0.042347077280282974, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.4857226014137268, "reward_total_composite_std": 0.12838341295719147} {"timestamp_utc": "2026-04-12T23:44:16Z", "mode": "train", "global_step": 632, "epoch": 0.0634856855851331, "loss": -0.0201, "grad_norm": 12.256442070007324, "learning_rate": 8.08787878787879e-06, "num_tokens": 1173110.0, "completions/mean_length": 42.625, "completions/min_length": 39.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.625, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.8699250817298889, "rewards/meter/std": 0.26429712772369385, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9893931746482849, "rewards/repeat_soft/std": 0.026070963591337204, "rewards/judge_quality/mean": 0.44624999165534973, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7742805480957031, "rewards/total_composite/std": 0.11792957782745361, "reward": 0.7742805480957031, "reward_std": 0.11792959272861481, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20290476083755493, "sampling/sampling_logp_difference/max": 1.6152982711791992, "sampling/importance_sampling_ratio/min": 0.19883136451244354, "sampling/importance_sampling_ratio/mean": 1.0324456691741943, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9458536803722382, "clip_ratio/low_mean": 0.03437500074505806, "clip_ratio/low_min": 0.03437500074505806, "clip_ratio/high_mean": 0.13397350441664457, "clip_ratio/high_max": 0.13397350441664457, "clip_ratio/region_mean": 0.16834850516170263, "reward_total_mean": 0.7742805480957031, "reward_meter_mean": 0.8699250817298889, "reward_meter_std": 0.26429712772369385, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9893931746482849, "reward_repeat_soft_std": 0.026070963591337204, "reward_judge_quality_mean": 0.44624999165534973, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7742805480957031, "reward_total_composite_std": 0.11792957782745361} {"timestamp_utc": "2026-04-12T23:44:23Z", "mode": "train", "global_step": 633, "epoch": 0.06358613761928679, "loss": -0.0126, "grad_norm": 11.208847045898438, "learning_rate": 8.084848484848485e-06, "num_tokens": 1175589.0, "completions/mean_length": 107.875, "completions/min_length": 83.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.875, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.9065282344818115, "rewards/meter/std": 0.18144384026527405, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.93458092212677, "rewards/repeat_soft/std": 0.09720819443464279, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.12351980805397034, "rewards/total_composite/mean": 0.7368957996368408, "rewards/total_composite/std": 0.1056220680475235, "reward": 0.7368957996368408, "reward_std": 0.1056220680475235, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18620237708091736, "sampling/sampling_logp_difference/max": 1.9081745147705078, "sampling/importance_sampling_ratio/min": 0.14835095405578613, "sampling/importance_sampling_ratio/mean": 1.0512454509735107, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8716929703950882, "clip_ratio/low_mean": 0.08224043622612953, "clip_ratio/low_min": 0.08224043622612953, "clip_ratio/high_mean": 0.08575570210814476, "clip_ratio/high_max": 0.08575570210814476, "clip_ratio/region_mean": 0.1679961383342743, "reward_total_mean": 0.7368957996368408, "reward_meter_mean": 0.9065282344818115, "reward_meter_std": 0.18144384026527405, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.93458092212677, "reward_repeat_soft_std": 0.09720819443464279, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.12351980805397034, "reward_total_composite_mean": 0.7368957996368408, "reward_total_composite_std": 0.1056220680475235} {"timestamp_utc": "2026-04-12T23:44:30Z", "mode": "train", "global_step": 634, "epoch": 0.06368658965344048, "loss": 0.0376, "grad_norm": 21.58633804321289, "learning_rate": 8.081818181818182e-06, "num_tokens": 1177124.0, "completions/mean_length": 40.875, "completions/min_length": 29.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.875, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.28774192929267883, "rewards/meter/std": 0.3885965645313263, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9616593718528748, "rewards/repeat_soft/std": 0.061550531536340714, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.5538997650146484, "rewards/total_composite/std": 0.1966932713985443, "reward": 0.5538997650146484, "reward_std": 0.1966932713985443, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1365634799003601, "sampling/sampling_logp_difference/max": 1.0540122985839844, "sampling/importance_sampling_ratio/min": 0.34853652119636536, "sampling/importance_sampling_ratio/mean": 1.0177440643310547, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8569293171167374, "clip_ratio/low_mean": 0.07444861903786659, "clip_ratio/low_min": 0.07444861903786659, "clip_ratio/high_mean": 0.03765528090298176, "clip_ratio/high_max": 0.03765528090298176, "clip_ratio/region_mean": 0.11210389994084835, "reward_total_mean": 0.5538997650146484, "reward_meter_mean": 0.28774192929267883, "reward_meter_std": 0.3885965645313263, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9616593718528748, "reward_repeat_soft_std": 0.061550531536340714, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.5538997650146484, "reward_total_composite_std": 0.1966932713985443} {"timestamp_utc": "2026-04-12T23:44:37Z", "mode": "train", "global_step": 635, "epoch": 0.06378704168759418, "loss": 0.0558, "grad_norm": 12.919676780700684, "learning_rate": 8.07878787878788e-06, "num_tokens": 1178870.0, "completions/mean_length": 58.25, "completions/min_length": 50.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.25, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.6197859048843384, "rewards/meter/std": 0.3277495205402374, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.986762285232544, "rewards/repeat_soft/std": 0.018921004608273506, "rewards/judge_quality/mean": 0.6325000524520874, "rewards/judge_quality/std": 0.23566018044948578, "rewards/total_composite/mean": 0.7173299193382263, "rewards/total_composite/std": 0.17934489250183105, "reward": 0.7173299193382263, "reward_std": 0.17934490740299225, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18254849314689636, "sampling/sampling_logp_difference/max": 1.452589988708496, "sampling/importance_sampling_ratio/min": 0.23396354913711548, "sampling/importance_sampling_ratio/mean": 1.0165400505065918, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4434458017349243, "clip_ratio/low_mean": 0.07037041988223791, "clip_ratio/low_min": 0.07037041988223791, "clip_ratio/high_mean": 0.08572741225361824, "clip_ratio/high_max": 0.08572741225361824, "clip_ratio/region_mean": 0.15609783213585615, "reward_total_mean": 0.7173299193382263, "reward_meter_mean": 0.6197859048843384, "reward_meter_std": 0.3277495205402374, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.986762285232544, "reward_repeat_soft_std": 0.018921004608273506, "reward_judge_quality_mean": 0.6325000524520874, "reward_judge_quality_std": 0.23566018044948578, "reward_total_composite_mean": 0.7173299193382263, "reward_total_composite_std": 0.17934489250183105} {"timestamp_utc": "2026-04-12T23:44:43Z", "mode": "train", "global_step": 636, "epoch": 0.06388749372174786, "loss": 0.0903, "grad_norm": 30.963809967041016, "learning_rate": 8.075757575757577e-06, "num_tokens": 1180750.0, "completions/mean_length": 60.0, "completions/min_length": 49.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.0, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.11805801093578339, "rewards/meter/std": 0.08673175424337387, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9523489475250244, "rewards/repeat_soft/std": 0.04631205275654793, "rewards/judge_quality/mean": 0.6825000047683716, "rewards/judge_quality/std": 0.23260945081710815, "rewards/total_composite/mean": 0.4984235167503357, "rewards/total_composite/std": 0.07696522027254105, "reward": 0.4984235167503357, "reward_std": 0.07696521282196045, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2054048776626587, "sampling/sampling_logp_difference/max": 2.2607078552246094, "sampling/importance_sampling_ratio/min": 0.104276642203331, "sampling/importance_sampling_ratio/mean": 1.0073902606964111, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0242095962166786, "clip_ratio/low_mean": 0.10127021186053753, "clip_ratio/low_min": 0.10127021186053753, "clip_ratio/high_mean": 0.07090557273477316, "clip_ratio/high_max": 0.07090557273477316, "clip_ratio/region_mean": 0.1721757845953107, "reward_total_mean": 0.4984235167503357, "reward_meter_mean": 0.11805801093578339, "reward_meter_std": 0.08673175424337387, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9523489475250244, "reward_repeat_soft_std": 0.04631205275654793, "reward_judge_quality_mean": 0.6825000047683716, "reward_judge_quality_std": 0.23260945081710815, "reward_total_composite_mean": 0.4984235167503357, "reward_total_composite_std": 0.07696522027254105} {"timestamp_utc": "2026-04-12T23:44:50Z", "mode": "train", "global_step": 637, "epoch": 0.06398794575590155, "loss": 0.0688, "grad_norm": 16.75688934326172, "learning_rate": 8.072727272727274e-06, "num_tokens": 1182439.0, "completions/mean_length": 36.125, "completions/min_length": 29.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.125, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.45947960019111633, "rewards/meter/std": 0.410409152507782, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9626700282096863, "rewards/repeat_soft/std": 0.08673539757728577, "rewards/judge_quality/mean": 0.6449999809265137, "rewards/judge_quality/std": 0.24928471446037292, "rewards/total_composite/mean": 0.6465327739715576, "rewards/total_composite/std": 0.14740228652954102, "reward": 0.6465327739715576, "reward_std": 0.1474023014307022, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16857413947582245, "sampling/sampling_logp_difference/max": 2.759812593460083, "sampling/importance_sampling_ratio/min": 0.0633036270737648, "sampling/importance_sampling_ratio/mean": 0.9991772174835205, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1212516501545906, "clip_ratio/low_mean": 0.08766834810376167, "clip_ratio/low_min": 0.08766834810376167, "clip_ratio/high_mean": 0.07121687941253185, "clip_ratio/high_max": 0.07121687941253185, "clip_ratio/region_mean": 0.15888522751629353, "reward_total_mean": 0.6465327739715576, "reward_meter_mean": 0.45947960019111633, "reward_meter_std": 0.410409152507782, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9626700282096863, "reward_repeat_soft_std": 0.08673539757728577, "reward_judge_quality_mean": 0.6449999809265137, "reward_judge_quality_std": 0.24928471446037292, "reward_total_composite_mean": 0.6465327739715576, "reward_total_composite_std": 0.14740228652954102} {"timestamp_utc": "2026-04-12T23:44:56Z", "mode": "train", "global_step": 638, "epoch": 0.06408839779005525, "loss": 0.0793, "grad_norm": 13.117921829223633, "learning_rate": 8.069696969696971e-06, "num_tokens": 1184546.0, "completions/mean_length": 70.375, "completions/min_length": 50.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.375, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.7077898979187012, "rewards/meter/std": 0.3417811691761017, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9676499962806702, "rewards/repeat_soft/std": 0.050677474588155746, "rewards/judge_quality/mean": 0.4737499952316284, "rewards/judge_quality/std": 0.16291432082653046, "rewards/total_composite/mean": 0.7027079463005066, "rewards/total_composite/std": 0.15831251442432404, "reward": 0.7027079463005066, "reward_std": 0.15831249952316284, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1735275834798813, "sampling/sampling_logp_difference/max": 3.9394397735595703, "sampling/importance_sampling_ratio/min": 0.019459113478660583, "sampling/importance_sampling_ratio/mean": 1.030045747756958, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4164995774626732, "clip_ratio/low_mean": 0.06599915213882923, "clip_ratio/low_min": 0.06599915213882923, "clip_ratio/high_mean": 0.09586472623050213, "clip_ratio/high_max": 0.09586472623050213, "clip_ratio/region_mean": 0.16186387836933136, "reward_total_mean": 0.7027079463005066, "reward_meter_mean": 0.7077898979187012, "reward_meter_std": 0.3417811691761017, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9676499962806702, "reward_repeat_soft_std": 0.050677474588155746, "reward_judge_quality_mean": 0.4737499952316284, "reward_judge_quality_std": 0.16291432082653046, "reward_total_composite_mean": 0.7027079463005066, "reward_total_composite_std": 0.15831251442432404} {"timestamp_utc": "2026-04-12T23:45:03Z", "mode": "train", "global_step": 639, "epoch": 0.06418884982420894, "loss": 0.0202, "grad_norm": 15.97309398651123, "learning_rate": 8.066666666666667e-06, "num_tokens": 1186283.0, "completions/mean_length": 45.125, "completions/min_length": 40.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.125, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.9260362386703491, "rewards/meter/std": 0.1685025542974472, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9755889177322388, "rewards/repeat_soft/std": 0.019224073737859726, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.8026502132415771, "rewards/total_composite/std": 0.08329853415489197, "reward": 0.8026502132415771, "reward_std": 0.08329854905605316, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17670723795890808, "sampling/sampling_logp_difference/max": 1.1942873001098633, "sampling/importance_sampling_ratio/min": 0.30291977524757385, "sampling/importance_sampling_ratio/mean": 1.028493046760559, "sampling/importance_sampling_ratio/max": 1.9228723049163818, "entropy": 1.8322037011384964, "clip_ratio/low_mean": 0.03055555559694767, "clip_ratio/low_min": 0.03055555559694767, "clip_ratio/high_mean": 0.1610116520896554, "clip_ratio/high_max": 0.1610116520896554, "clip_ratio/region_mean": 0.19156720768660307, "reward_total_mean": 0.8026502132415771, "reward_meter_mean": 0.9260362386703491, "reward_meter_std": 0.1685025542974472, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9755889177322388, "reward_repeat_soft_std": 0.019224073737859726, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.8026502132415771, "reward_total_composite_std": 0.08329853415489197} {"timestamp_utc": "2026-04-12T23:45:11Z", "mode": "train", "global_step": 640, "epoch": 0.06428930185836264, "loss": 0.0038, "grad_norm": 9.011969566345215, "learning_rate": 8.063636363636364e-06, "num_tokens": 1188810.0, "completions/mean_length": 112.875, "completions/min_length": 86.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.875, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.9031827449798584, "rewards/meter/std": 0.17064249515533447, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9518580436706543, "rewards/repeat_soft/std": 0.030199479311704636, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7547430396080017, "rewards/total_composite/std": 0.09310760349035263, "reward": 0.7547430396080017, "reward_std": 0.09310761094093323, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17082273960113525, "sampling/sampling_logp_difference/max": 1.3971529006958008, "sampling/importance_sampling_ratio/min": 0.24730005860328674, "sampling/importance_sampling_ratio/mean": 1.046204924583435, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6927846521139145, "clip_ratio/low_mean": 0.048563042655587196, "clip_ratio/low_min": 0.048563042655587196, "clip_ratio/high_mean": 0.12603239342570305, "clip_ratio/high_max": 0.12603239342570305, "clip_ratio/region_mean": 0.17459543608129025, "reward_total_mean": 0.7547430396080017, "reward_meter_mean": 0.9031827449798584, "reward_meter_std": 0.17064249515533447, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9518580436706543, "reward_repeat_soft_std": 0.030199479311704636, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7547430396080017, "reward_total_composite_std": 0.09310760349035263} {"timestamp_utc": "2026-04-12T23:45:16Z", "mode": "train", "global_step": 641, "epoch": 0.06438975389251632, "loss": -0.0213, "grad_norm": 20.656909942626953, "learning_rate": 8.060606060606061e-06, "num_tokens": 1190418.0, "completions/mean_length": 30.0, "completions/min_length": 27.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.0, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.288332462310791, "rewards/meter/std": 0.36374345421791077, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9806827306747437, "rewards/repeat_soft/std": 0.010548335500061512, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.4764636158943176, "rewards/total_composite/std": 0.24179093539714813, "reward": 0.4764636158943176, "reward_std": 0.24179092049598694, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14092130959033966, "sampling/sampling_logp_difference/max": 1.3633909225463867, "sampling/importance_sampling_ratio/min": 0.25579193234443665, "sampling/importance_sampling_ratio/mean": 1.0030680894851685, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8778230100870132, "clip_ratio/low_mean": 0.0553759578615427, "clip_ratio/low_min": 0.0553759578615427, "clip_ratio/high_mean": 0.06808601692318916, "clip_ratio/high_max": 0.06808601692318916, "clip_ratio/region_mean": 0.12346197478473186, "reward_total_mean": 0.4764636158943176, "reward_meter_mean": 0.288332462310791, "reward_meter_std": 0.36374345421791077, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9806827306747437, "reward_repeat_soft_std": 0.010548335500061512, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.4764636158943176, "reward_total_composite_std": 0.24179093539714813} {"timestamp_utc": "2026-04-12T23:45:25Z", "mode": "train", "global_step": 642, "epoch": 0.06449020592667001, "loss": 0.0095, "grad_norm": 14.173985481262207, "learning_rate": 8.057575757575759e-06, "num_tokens": 1192313.0, "completions/mean_length": 71.875, "completions/min_length": 65.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.875, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.6396819949150085, "rewards/meter/std": 0.38585105538368225, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9644151926040649, "rewards/repeat_soft/std": 0.03777209296822548, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.6099046468734741, "rewards/total_composite/std": 0.2842549681663513, "reward": 0.6099046468734741, "reward_std": 0.2842549681663513, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1921178698539734, "sampling/sampling_logp_difference/max": 2.0746877193450928, "sampling/importance_sampling_ratio/min": 0.12559564411640167, "sampling/importance_sampling_ratio/mean": 1.0236910581588745, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1294835433363914, "clip_ratio/low_mean": 0.06805377639830112, "clip_ratio/low_min": 0.06805377639830112, "clip_ratio/high_mean": 0.1037131454795599, "clip_ratio/high_max": 0.1037131454795599, "clip_ratio/region_mean": 0.17176692187786102, "reward_total_mean": 0.6099046468734741, "reward_meter_mean": 0.6396819949150085, "reward_meter_std": 0.38585105538368225, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9644151926040649, "reward_repeat_soft_std": 0.03777209296822548, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.6099046468734741, "reward_total_composite_std": 0.2842549681663513} {"timestamp_utc": "2026-04-12T23:45:31Z", "mode": "train", "global_step": 643, "epoch": 0.06459065796082371, "loss": -0.0281, "grad_norm": 25.883132934570312, "learning_rate": 8.054545454545454e-06, "num_tokens": 1193725.0, "completions/mean_length": 19.5, "completions/min_length": 16.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.5, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.36635541915893555, "rewards/meter/std": 0.3399078845977783, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9196497201919556, "rewards/repeat_soft/std": 0.05091951787471771, "rewards/judge_quality/mean": 0.3424999713897705, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.5095748901367188, "rewards/total_composite/std": 0.1759835034608841, "reward": 0.5095748901367188, "reward_std": 0.1759834885597229, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20458479225635529, "sampling/sampling_logp_difference/max": 1.6669042110443115, "sampling/importance_sampling_ratio/min": 0.18883073329925537, "sampling/importance_sampling_ratio/mean": 1.0277245044708252, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5964243113994598, "clip_ratio/low_mean": 0.1348329484462738, "clip_ratio/low_min": 0.1348329484462738, "clip_ratio/high_mean": 0.04156746156513691, "clip_ratio/high_max": 0.04156746156513691, "clip_ratio/region_mean": 0.1764004100114107, "reward_total_mean": 0.5095748901367188, "reward_meter_mean": 0.36635541915893555, "reward_meter_std": 0.3399078845977783, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9196497201919556, "reward_repeat_soft_std": 0.05091951787471771, "reward_judge_quality_mean": 0.3424999713897705, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.5095748901367188, "reward_total_composite_std": 0.1759835034608841} {"timestamp_utc": "2026-04-12T23:45:38Z", "mode": "train", "global_step": 644, "epoch": 0.0646911099949774, "loss": 0.0484, "grad_norm": 16.172021865844727, "learning_rate": 8.051515151515153e-06, "num_tokens": 1195922.0, "completions/mean_length": 86.625, "completions/min_length": 77.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.625, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.6857371926307678, "rewards/meter/std": 0.27419325709342957, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9857232570648193, "rewards/repeat_soft/std": 0.00756278820335865, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.6955291032791138, "rewards/total_composite/std": 0.13680893182754517, "reward": 0.6955291032791138, "reward_std": 0.13680893182754517, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2439299076795578, "sampling/sampling_logp_difference/max": 2.37593412399292, "sampling/importance_sampling_ratio/min": 0.09292764216661453, "sampling/importance_sampling_ratio/mean": 1.0161646604537964, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4725598692893982, "clip_ratio/low_mean": 0.11439409479498863, "clip_ratio/low_min": 0.11439409479498863, "clip_ratio/high_mean": 0.10690774209797382, "clip_ratio/high_max": 0.10690774209797382, "clip_ratio/region_mean": 0.22130183689296246, "reward_total_mean": 0.6955291032791138, "reward_meter_mean": 0.6857371926307678, "reward_meter_std": 0.27419325709342957, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9857232570648193, "reward_repeat_soft_std": 0.00756278820335865, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.6955291032791138, "reward_total_composite_std": 0.13680893182754517} {"timestamp_utc": "2026-04-12T23:45:44Z", "mode": "train", "global_step": 645, "epoch": 0.06479156202913108, "loss": -0.049, "grad_norm": 14.340354919433594, "learning_rate": 8.048484848484849e-06, "num_tokens": 1197620.0, "completions/mean_length": 46.25, "completions/min_length": 33.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.25, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.593339204788208, "rewards/meter/std": 0.33179134130477905, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9698635935783386, "rewards/repeat_soft/std": 0.022010376676917076, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.6699889898300171, "rewards/total_composite/std": 0.14403800666332245, "reward": 0.6699889898300171, "reward_std": 0.14403800666332245, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1834997534751892, "sampling/sampling_logp_difference/max": 1.8247432708740234, "sampling/importance_sampling_ratio/min": 0.16125904023647308, "sampling/importance_sampling_ratio/mean": 1.0336220264434814, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4576058089733124, "clip_ratio/low_mean": 0.08540664613246918, "clip_ratio/low_min": 0.08540664613246918, "clip_ratio/high_mean": 0.10028313286602497, "clip_ratio/high_max": 0.10028313286602497, "clip_ratio/region_mean": 0.18568977899849415, "reward_total_mean": 0.6699889898300171, "reward_meter_mean": 0.593339204788208, "reward_meter_std": 0.33179134130477905, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9698635935783386, "reward_repeat_soft_std": 0.022010376676917076, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.6699889898300171, "reward_total_composite_std": 0.14403800666332245} {"timestamp_utc": "2026-04-12T23:45:50Z", "mode": "train", "global_step": 646, "epoch": 0.06489201406328478, "loss": -0.0074, "grad_norm": 14.02996826171875, "learning_rate": 8.045454545454546e-06, "num_tokens": 1199088.0, "completions/mean_length": 38.5, "completions/min_length": 35.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.5, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.6672691106796265, "rewards/meter/std": 0.4244491755962372, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9995162487030029, "rewards/repeat_soft/std": 0.0013681710697710514, "rewards/judge_quality/mean": 0.815000057220459, "rewards/judge_quality/std": 0.16970564424991608, "rewards/total_composite/mean": 0.7947227358818054, "rewards/total_composite/std": 0.20978976786136627, "reward": 0.7947227358818054, "reward_std": 0.20978976786136627, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16091148555278778, "sampling/sampling_logp_difference/max": 7.138713836669922, "sampling/importance_sampling_ratio/min": 0.0007937724003568292, "sampling/importance_sampling_ratio/mean": 1.0003782510757446, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1334645673632622, "clip_ratio/low_mean": 0.06933740712702274, "clip_ratio/low_min": 0.06933740712702274, "clip_ratio/high_mean": 0.08459823857992887, "clip_ratio/high_max": 0.08459823857992887, "clip_ratio/region_mean": 0.15393564570695162, "reward_total_mean": 0.7947227358818054, "reward_meter_mean": 0.6672691106796265, "reward_meter_std": 0.4244491755962372, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9995162487030029, "reward_repeat_soft_std": 0.0013681710697710514, "reward_judge_quality_mean": 0.815000057220459, "reward_judge_quality_std": 0.16970564424991608, "reward_total_composite_mean": 0.7947227358818054, "reward_total_composite_std": 0.20978976786136627} {"timestamp_utc": "2026-04-12T23:45:57Z", "mode": "train", "global_step": 647, "epoch": 0.06499246609743847, "loss": -0.0007, "grad_norm": 11.056327819824219, "learning_rate": 8.042424242424243e-06, "num_tokens": 1201161.0, "completions/mean_length": 66.125, "completions/min_length": 51.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.125, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.6852612495422363, "rewards/meter/std": 0.32589441537857056, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9734658002853394, "rewards/repeat_soft/std": 0.029375718906521797, "rewards/judge_quality/mean": 0.6187499761581421, "rewards/judge_quality/std": 0.21596544981002808, "rewards/total_composite/mean": 0.6235469579696655, "rewards/total_composite/std": 0.3003011643886566, "reward": 0.6235469579696655, "reward_std": 0.3003011643886566, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1811431348323822, "sampling/sampling_logp_difference/max": 1.6939148902893066, "sampling/importance_sampling_ratio/min": 0.18379856646060944, "sampling/importance_sampling_ratio/mean": 1.0138866901397705, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4475714266300201, "clip_ratio/low_mean": 0.09227598831057549, "clip_ratio/low_min": 0.09227598831057549, "clip_ratio/high_mean": 0.08282938972115517, "clip_ratio/high_max": 0.08282938972115517, "clip_ratio/region_mean": 0.17510537803173065, "reward_total_mean": 0.6235469579696655, "reward_meter_mean": 0.6852612495422363, "reward_meter_std": 0.32589441537857056, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9734658002853394, "reward_repeat_soft_std": 0.029375718906521797, "reward_judge_quality_mean": 0.6187499761581421, "reward_judge_quality_std": 0.21596544981002808, "reward_total_composite_mean": 0.6235469579696655, "reward_total_composite_std": 0.3003011643886566} {"timestamp_utc": "2026-04-12T23:46:04Z", "mode": "train", "global_step": 648, "epoch": 0.06509291813159217, "loss": 0.0531, "grad_norm": 15.861492156982422, "learning_rate": 8.03939393939394e-06, "num_tokens": 1202639.0, "completions/mean_length": 41.75, "completions/min_length": 35.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.75, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.7381439208984375, "rewards/meter/std": 0.3066772520542145, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9900707006454468, "rewards/repeat_soft/std": 0.011574761010706425, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.7195468544960022, "rewards/total_composite/std": 0.12471076846122742, "reward": 0.7195468544960022, "reward_std": 0.12471075356006622, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18581396341323853, "sampling/sampling_logp_difference/max": 2.004092216491699, "sampling/importance_sampling_ratio/min": 0.1347825825214386, "sampling/importance_sampling_ratio/mean": 1.035754919052124, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5621098279953003, "clip_ratio/low_mean": 0.06063830014318228, "clip_ratio/low_min": 0.06063830014318228, "clip_ratio/high_mean": 0.12428988702595234, "clip_ratio/high_max": 0.12428988702595234, "clip_ratio/region_mean": 0.18492818716913462, "reward_total_mean": 0.7195468544960022, "reward_meter_mean": 0.7381439208984375, "reward_meter_std": 0.3066772520542145, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9900707006454468, "reward_repeat_soft_std": 0.011574761010706425, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.7195468544960022, "reward_total_composite_std": 0.12471076846122742} {"timestamp_utc": "2026-04-12T23:46:11Z", "mode": "train", "global_step": 649, "epoch": 0.06519337016574586, "loss": 0.0417, "grad_norm": 12.813417434692383, "learning_rate": 8.036363636363636e-06, "num_tokens": 1204404.0, "completions/mean_length": 41.625, "completions/min_length": 34.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.625, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.7462376356124878, "rewards/meter/std": 0.2753569483757019, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9860265254974365, "rewards/repeat_soft/std": 0.02116462029516697, "rewards/judge_quality/mean": 0.6725000143051147, "rewards/judge_quality/std": 0.2544882893562317, "rewards/total_composite/mean": 0.6857874393463135, "rewards/total_composite/std": 0.31209564208984375, "reward": 0.6857874393463135, "reward_std": 0.31209564208984375, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1689222753047943, "sampling/sampling_logp_difference/max": 1.5345954895019531, "sampling/importance_sampling_ratio/min": 0.21554288268089294, "sampling/importance_sampling_ratio/mean": 1.0316044092178345, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.611610285937786, "clip_ratio/low_mean": 0.05610877927392721, "clip_ratio/low_min": 0.05610877927392721, "clip_ratio/high_mean": 0.11560218594968319, "clip_ratio/high_max": 0.11560218594968319, "clip_ratio/region_mean": 0.1717109652236104, "reward_total_mean": 0.6857874393463135, "reward_meter_mean": 0.7462376356124878, "reward_meter_std": 0.2753569483757019, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9860265254974365, "reward_repeat_soft_std": 0.02116462029516697, "reward_judge_quality_mean": 0.6725000143051147, "reward_judge_quality_std": 0.2544882893562317, "reward_total_composite_mean": 0.6857874393463135, "reward_total_composite_std": 0.31209564208984375} {"timestamp_utc": "2026-04-12T23:46:19Z", "mode": "train", "global_step": 650, "epoch": 0.06529382219989954, "loss": -0.0204, "grad_norm": 12.047786712646484, "learning_rate": 8.033333333333335e-06, "num_tokens": 1206278.0, "completions/mean_length": 70.25, "completions/min_length": 54.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.25, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.93299800157547, "rewards/meter/std": 0.10141206532716751, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9897191524505615, "rewards/repeat_soft/std": 0.013279805891215801, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.7920085191726685, "rewards/total_composite/std": 0.0602944940328598, "reward": 0.7920085191726685, "reward_std": 0.0602944977581501, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19080102443695068, "sampling/sampling_logp_difference/max": 1.479166030883789, "sampling/importance_sampling_ratio/min": 0.22782762348651886, "sampling/importance_sampling_ratio/mean": 1.020076870918274, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7764126509428024, "clip_ratio/low_mean": 0.09778391011059284, "clip_ratio/low_min": 0.09778391011059284, "clip_ratio/high_mean": 0.10057042352855206, "clip_ratio/high_max": 0.10057042352855206, "clip_ratio/region_mean": 0.1983543336391449, "reward_total_mean": 0.7920085191726685, "reward_meter_mean": 0.93299800157547, "reward_meter_std": 0.10141206532716751, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9897191524505615, "reward_repeat_soft_std": 0.013279805891215801, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.7920085191726685, "reward_total_composite_std": 0.0602944940328598} {"timestamp_utc": "2026-04-12T23:47:05Z", "mode": "eval", "global_step": 650, "epoch": 0.06529382219989954, "eval_loss": NaN, "eval_runtime": 46.3349, "eval_samples_per_second": 1.727, "eval_steps_per_second": 0.216, "eval_num_tokens": 1206278.0, "eval_completions/mean_length": 70.925, "eval_completions/min_length": 28.1, "eval_completions/max_length": 159.6, "eval_completions/clipped_ratio": 0.0125, "eval_completions/mean_terminated_length": 65.25357170104981, "eval_completions/min_terminated_length": 28.1, "eval_completions/max_terminated_length": 117.6, "eval_rewards/meter/mean": 0.6953476667404175, "eval_rewards/meter/std": 0.3057481192052364, "eval_rewards/count_adherence/mean": 0.9675000011920929, "eval_rewards/count_adherence/std": 0.08703994564712048, "eval_rewards/hard_gate/mean": 0.9625, "eval_rewards/hard_gate/std": 0.0816463440656662, "eval_rewards/repeat_soft/mean": 0.9802694499492646, "eval_rewards/repeat_soft/std": 0.023311478924006222, "eval_rewards/judge_quality/mean": 0.4366249978542328, "eval_rewards/judge_quality/std": 0.1384577125310898, "eval_rewards/total_composite/mean": 0.6665221095085144, "eval_rewards/total_composite/std": 0.1804937306791544, "eval_reward": 0.6665221095085144, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.12736358493566513, "eval_sampling/sampling_logp_difference/max": 1.1890414714813233, "eval_sampling/importance_sampling_ratio/min": 0.316360330581665, "eval_sampling/importance_sampling_ratio/mean": 1.037365233898163, "eval_sampling/importance_sampling_ratio/max": 1.5868561625480653, "eval_entropy": 1.692005729675293, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6665221095085144, "eval_reward_meter_mean": 0.6953476667404175, "eval_reward_meter_std": 0.3057481192052364, "eval_reward_count_adherence_mean": 0.9675000011920929, "eval_reward_count_adherence_std": 0.08703994564712048, "eval_reward_hard_gate_mean": 0.9625, "eval_reward_hard_gate_std": 0.0816463440656662, "eval_reward_repeat_soft_mean": 0.9802694499492646, "eval_reward_repeat_soft_std": 0.023311478924006222, "eval_reward_judge_quality_mean": 0.4366249978542328, "eval_reward_judge_quality_std": 0.1384577125310898, "eval_reward_total_composite_mean": 0.6665221095085144, "eval_reward_total_composite_std": 0.1804937306791544} {"timestamp_utc": "2026-04-12T23:47:14Z", "mode": "train", "global_step": 651, "epoch": 0.06539427423405324, "loss": 0.0785, "grad_norm": 13.712069511413574, "learning_rate": 8.03030303030303e-06, "num_tokens": 1208354.0, "completions/mean_length": 66.5, "completions/min_length": 52.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9185701608657837, "rewards/meter/std": 0.15826614201068878, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9620794057846069, "rewards/repeat_soft/std": 0.03613687679171562, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.7626895308494568, "rewards/total_composite/std": 0.06990593671798706, "reward": 0.7626895308494568, "reward_std": 0.06990595161914825, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1786600798368454, "sampling/sampling_logp_difference/max": 2.4785397052764893, "sampling/importance_sampling_ratio/min": 0.08386560529470444, "sampling/importance_sampling_ratio/mean": 1.0065163373947144, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3620951026678085, "clip_ratio/low_mean": 0.054327878169715405, "clip_ratio/low_min": 0.054327878169715405, "clip_ratio/high_mean": 0.08453443367034197, "clip_ratio/high_max": 0.08453443367034197, "clip_ratio/region_mean": 0.13886231184005737, "reward_total_mean": 0.7626895308494568, "reward_meter_mean": 0.9185701608657837, "reward_meter_std": 0.15826614201068878, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9620794057846069, "reward_repeat_soft_std": 0.03613687679171562, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.7626895308494568, "reward_total_composite_std": 0.06990593671798706} {"timestamp_utc": "2026-04-12T23:47:21Z", "mode": "train", "global_step": 652, "epoch": 0.06549472626820693, "loss": 0.0549, "grad_norm": 13.802163124084473, "learning_rate": 8.027272727272728e-06, "num_tokens": 1210166.0, "completions/mean_length": 65.5, "completions/min_length": 54.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.5, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.8982151746749878, "rewards/meter/std": 0.12974600493907928, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9283890724182129, "rewards/repeat_soft/std": 0.07566359639167786, "rewards/judge_quality/mean": 0.6112499833106995, "rewards/judge_quality/std": 0.15869447588920593, "rewards/total_composite/mean": 0.8304107189178467, "rewards/total_composite/std": 0.09198489785194397, "reward": 0.8304107189178467, "reward_std": 0.09198489040136337, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18080532550811768, "sampling/sampling_logp_difference/max": 2.392390251159668, "sampling/importance_sampling_ratio/min": 0.09141092747449875, "sampling/importance_sampling_ratio/mean": 1.0293105840682983, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5506623014807701, "clip_ratio/low_mean": 0.06523305457085371, "clip_ratio/low_min": 0.06523305457085371, "clip_ratio/high_mean": 0.1162495706230402, "clip_ratio/high_max": 0.1162495706230402, "clip_ratio/region_mean": 0.1814826251938939, "reward_total_mean": 0.8304107189178467, "reward_meter_mean": 0.8982151746749878, "reward_meter_std": 0.12974600493907928, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9283890724182129, "reward_repeat_soft_std": 0.07566359639167786, "reward_judge_quality_mean": 0.6112499833106995, "reward_judge_quality_std": 0.15869447588920593, "reward_total_composite_mean": 0.8304107189178467, "reward_total_composite_std": 0.09198489785194397} {"timestamp_utc": "2026-04-12T23:47:27Z", "mode": "train", "global_step": 653, "epoch": 0.06559517830236063, "loss": 0.1015, "grad_norm": 12.957351684570312, "learning_rate": 8.024242424242425e-06, "num_tokens": 1211882.0, "completions/mean_length": 48.5, "completions/min_length": 39.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.5, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.8946439623832703, "rewards/meter/std": 0.13428914546966553, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9610128402709961, "rewards/repeat_soft/std": 0.07620740681886673, "rewards/judge_quality/mean": 0.6987500190734863, "rewards/judge_quality/std": 0.3165184557437897, "rewards/total_composite/mean": 0.8583160638809204, "rewards/total_composite/std": 0.11272698640823364, "reward": 0.8583160638809204, "reward_std": 0.11272697895765305, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19252219796180725, "sampling/sampling_logp_difference/max": 1.4298181533813477, "sampling/importance_sampling_ratio/min": 0.23935244977474213, "sampling/importance_sampling_ratio/mean": 1.0273962020874023, "sampling/importance_sampling_ratio/max": 1.8949954509735107, "entropy": 1.8653132617473602, "clip_ratio/low_mean": 0.06383458944037557, "clip_ratio/low_min": 0.06383458944037557, "clip_ratio/high_mean": 0.08808714803308249, "clip_ratio/high_max": 0.08808714803308249, "clip_ratio/region_mean": 0.15192173747345805, "reward_total_mean": 0.8583160638809204, "reward_meter_mean": 0.8946439623832703, "reward_meter_std": 0.13428914546966553, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9610128402709961, "reward_repeat_soft_std": 0.07620740681886673, "reward_judge_quality_mean": 0.6987500190734863, "reward_judge_quality_std": 0.3165184557437897, "reward_total_composite_mean": 0.8583160638809204, "reward_total_composite_std": 0.11272698640823364} {"timestamp_utc": "2026-04-12T23:47:39Z", "mode": "train", "global_step": 654, "epoch": 0.06569563033651432, "loss": -0.1076, "grad_norm": 4.195548057556152, "learning_rate": 8.021212121212122e-06, "num_tokens": 1213687.0, "completions/mean_length": 115.625, "completions/min_length": 47.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 59.000003814697266, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.6900915503501892, "rewards/meter/std": 0.41646164655685425, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9167414307594299, "rewards/repeat_soft/std": 0.033695872873067856, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.60737144947052, "rewards/total_composite/std": 0.28649207949638367, "reward": 0.60737144947052, "reward_std": 0.28649207949638367, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18151891231536865, "sampling/sampling_logp_difference/max": 4.087885856628418, "sampling/importance_sampling_ratio/min": 0.016774659976363182, "sampling/importance_sampling_ratio/mean": 1.0177229642868042, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3013242185115814, "clip_ratio/low_mean": 0.01584506966173649, "clip_ratio/low_min": 0.01584506966173649, "clip_ratio/high_mean": 0.13556820340454578, "clip_ratio/high_max": 0.13556820340454578, "clip_ratio/region_mean": 0.15141327306628227, "reward_total_mean": 0.60737144947052, "reward_meter_mean": 0.6900915503501892, "reward_meter_std": 0.41646164655685425, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9167414307594299, "reward_repeat_soft_std": 0.033695872873067856, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.60737144947052, "reward_total_composite_std": 0.28649207949638367} {"timestamp_utc": "2026-04-12T23:47:51Z", "mode": "train", "global_step": 655, "epoch": 0.065796082370668, "loss": 0.0757, "grad_norm": 13.823602676391602, "learning_rate": 8.018181818181818e-06, "num_tokens": 1215462.0, "completions/mean_length": 54.875, "completions/min_length": 46.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.875, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.8082154393196106, "rewards/meter/std": 0.3065265417098999, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9780839085578918, "rewards/repeat_soft/std": 0.014694217592477798, "rewards/judge_quality/mean": 0.35999998450279236, "rewards/judge_quality/std": 0.09165150672197342, "rewards/total_composite/mean": 0.7195053696632385, "rewards/total_composite/std": 0.1569550484418869, "reward": 0.7195053696632385, "reward_std": 0.1569550335407257, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1983339637517929, "sampling/sampling_logp_difference/max": 1.239588737487793, "sampling/importance_sampling_ratio/min": 0.289503276348114, "sampling/importance_sampling_ratio/mean": 1.0260517597198486, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.055840477347374, "clip_ratio/low_mean": 0.05681438185274601, "clip_ratio/low_min": 0.05681438185274601, "clip_ratio/high_mean": 0.15467693470418453, "clip_ratio/high_max": 0.15467693470418453, "clip_ratio/region_mean": 0.21149131655693054, "reward_total_mean": 0.7195053696632385, "reward_meter_mean": 0.8082154393196106, "reward_meter_std": 0.3065265417098999, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9780839085578918, "reward_repeat_soft_std": 0.014694217592477798, "reward_judge_quality_mean": 0.35999998450279236, "reward_judge_quality_std": 0.09165150672197342, "reward_total_composite_mean": 0.7195053696632385, "reward_total_composite_std": 0.1569550484418869} {"timestamp_utc": "2026-04-12T23:47:59Z", "mode": "train", "global_step": 656, "epoch": 0.0658965344048217, "loss": 0.0033, "grad_norm": 9.732645988464355, "learning_rate": 8.015151515151515e-06, "num_tokens": 1217639.0, "completions/mean_length": 84.125, "completions/min_length": 70.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.125, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.7183910608291626, "rewards/meter/std": 0.35498619079589844, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9468309283256531, "rewards/repeat_soft/std": 0.05978486314415932, "rewards/judge_quality/mean": 0.3137499690055847, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.6620841026306152, "rewards/total_composite/std": 0.15048719942569733, "reward": 0.6620841026306152, "reward_std": 0.15048719942569733, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21116317808628082, "sampling/sampling_logp_difference/max": 1.587850570678711, "sampling/importance_sampling_ratio/min": 0.20436438918113708, "sampling/importance_sampling_ratio/mean": 1.0484699010849, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.220294713973999, "clip_ratio/low_mean": 0.05215000733733177, "clip_ratio/low_min": 0.05215000733733177, "clip_ratio/high_mean": 0.10843546781688929, "clip_ratio/high_max": 0.10843546781688929, "clip_ratio/region_mean": 0.16058547515422106, "reward_total_mean": 0.6620841026306152, "reward_meter_mean": 0.7183910608291626, "reward_meter_std": 0.35498619079589844, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9468309283256531, "reward_repeat_soft_std": 0.05978486314415932, "reward_judge_quality_mean": 0.3137499690055847, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.6620841026306152, "reward_total_composite_std": 0.15048719942569733} {"timestamp_utc": "2026-04-12T23:48:06Z", "mode": "train", "global_step": 657, "epoch": 0.06599698643897539, "loss": 0.0677, "grad_norm": 10.665203094482422, "learning_rate": 8.012121212121214e-06, "num_tokens": 1219568.0, "completions/mean_length": 66.125, "completions/min_length": 52.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.125, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.6103582978248596, "rewards/meter/std": 0.3560350835323334, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9794741272926331, "rewards/repeat_soft/std": 0.010189371183514595, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.6328586339950562, "rewards/total_composite/std": 0.16072311997413635, "reward": 0.6328586339950562, "reward_std": 0.16072311997413635, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18976569175720215, "sampling/sampling_logp_difference/max": 1.9324445724487305, "sampling/importance_sampling_ratio/min": 0.1447938084602356, "sampling/importance_sampling_ratio/mean": 1.0230695009231567, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7105656862258911, "clip_ratio/low_mean": 0.0893539022654295, "clip_ratio/low_min": 0.0893539022654295, "clip_ratio/high_mean": 0.09027033671736717, "clip_ratio/high_max": 0.09027033671736717, "clip_ratio/region_mean": 0.17962423898279667, "reward_total_mean": 0.6328586339950562, "reward_meter_mean": 0.6103582978248596, "reward_meter_std": 0.3560350835323334, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9794741272926331, "reward_repeat_soft_std": 0.010189371183514595, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.6328586339950562, "reward_total_composite_std": 0.16072311997413635} {"timestamp_utc": "2026-04-12T23:48:12Z", "mode": "train", "global_step": 658, "epoch": 0.06609743847312909, "loss": 0.0154, "grad_norm": 14.509693145751953, "learning_rate": 8.00909090909091e-06, "num_tokens": 1221245.0, "completions/mean_length": 38.625, "completions/min_length": 35.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.625, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.583063542842865, "rewards/meter/std": 0.41245514154434204, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9932006597518921, "rewards/repeat_soft/std": 0.00854469183832407, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.6376986503601074, "rewards/total_composite/std": 0.18562209606170654, "reward": 0.6376986503601074, "reward_std": 0.18562209606170654, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17925359308719635, "sampling/sampling_logp_difference/max": 1.6235942840576172, "sampling/importance_sampling_ratio/min": 0.19718867540359497, "sampling/importance_sampling_ratio/mean": 1.0705652236938477, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.689973622560501, "clip_ratio/low_mean": 0.05210116319358349, "clip_ratio/low_min": 0.05210116319358349, "clip_ratio/high_mean": 0.10056917648762465, "clip_ratio/high_max": 0.10056917648762465, "clip_ratio/region_mean": 0.15267033968120813, "reward_total_mean": 0.6376986503601074, "reward_meter_mean": 0.583063542842865, "reward_meter_std": 0.41245514154434204, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9932006597518921, "reward_repeat_soft_std": 0.00854469183832407, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.6376986503601074, "reward_total_composite_std": 0.18562209606170654} {"timestamp_utc": "2026-04-12T23:48:18Z", "mode": "train", "global_step": 659, "epoch": 0.06619789050728277, "loss": -0.0171, "grad_norm": 15.094762802124023, "learning_rate": 8.006060606060607e-06, "num_tokens": 1222826.0, "completions/mean_length": 40.625, "completions/min_length": 37.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.625, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.8171103000640869, "rewards/meter/std": 0.338232159614563, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9673138856887817, "rewards/repeat_soft/std": 0.036108314990997314, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.1011011004447937, "rewards/total_composite/mean": 0.6690179705619812, "rewards/total_composite/std": 0.31372541189193726, "reward": 0.6690179705619812, "reward_std": 0.31372538208961487, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22816431522369385, "sampling/sampling_logp_difference/max": 1.6073949337005615, "sampling/importance_sampling_ratio/min": 0.20040901005268097, "sampling/importance_sampling_ratio/mean": 1.0531185865402222, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6089470386505127, "clip_ratio/low_mean": 0.030405404046177864, "clip_ratio/low_min": 0.030405404046177864, "clip_ratio/high_mean": 0.15439960034564137, "clip_ratio/high_max": 0.15439960034564137, "clip_ratio/region_mean": 0.18480500439181924, "reward_total_mean": 0.6690179705619812, "reward_meter_mean": 0.8171103000640869, "reward_meter_std": 0.338232159614563, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9673138856887817, "reward_repeat_soft_std": 0.036108314990997314, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.1011011004447937, "reward_total_composite_mean": 0.6690179705619812, "reward_total_composite_std": 0.31372541189193726} {"timestamp_utc": "2026-04-12T23:48:25Z", "mode": "train", "global_step": 660, "epoch": 0.06629834254143646, "loss": 0.0374, "grad_norm": 17.13682746887207, "learning_rate": 8.003030303030304e-06, "num_tokens": 1224418.0, "completions/mean_length": 32.0, "completions/min_length": 26.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9554543495178223, "rewards/meter/std": 0.07299633324146271, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9758310914039612, "rewards/repeat_soft/std": 0.015864983201026917, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.8597875833511353, "rewards/total_composite/std": 0.09213182330131531, "reward": 0.8597875833511353, "reward_std": 0.0921318382024765, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12970466911792755, "sampling/sampling_logp_difference/max": 0.9377851486206055, "sampling/importance_sampling_ratio/min": 0.39272448420524597, "sampling/importance_sampling_ratio/mean": 1.0159516334533691, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8535038530826569, "clip_ratio/low_mean": 0.09498812351375818, "clip_ratio/low_min": 0.09498812351375818, "clip_ratio/high_mean": 0.042700715363025665, "clip_ratio/high_max": 0.042700715363025665, "clip_ratio/region_mean": 0.13768883887678385, "reward_total_mean": 0.8597875833511353, "reward_meter_mean": 0.9554543495178223, "reward_meter_std": 0.07299633324146271, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9758310914039612, "reward_repeat_soft_std": 0.015864983201026917, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.8597875833511353, "reward_total_composite_std": 0.09213182330131531} {"timestamp_utc": "2026-04-12T23:48:36Z", "mode": "train", "global_step": 661, "epoch": 0.06639879457559016, "loss": -0.0729, "grad_norm": 3.051436185836792, "learning_rate": 8.000000000000001e-06, "num_tokens": 1225694.0, "completions/mean_length": 84.5, "completions/min_length": 19.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 23.428571701049805, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.8658849000930786, "rewards/meter/std": 0.2432955801486969, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.45125001668930054, "rewards/judge_quality/std": 0.23381541669368744, "rewards/total_composite/mean": 0.6913444399833679, "rewards/total_composite/std": 0.30673715472221375, "reward": 0.6913444399833679, "reward_std": 0.30673715472221375, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18591085076332092, "sampling/sampling_logp_difference/max": 1.8631601333618164, "sampling/importance_sampling_ratio/min": 0.15518146753311157, "sampling/importance_sampling_ratio/mean": 1.0453687906265259, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3382781818509102, "clip_ratio/low_mean": 0.02083333395421505, "clip_ratio/low_min": 0.02083333395421505, "clip_ratio/high_mean": 0.12051025498658419, "clip_ratio/high_max": 0.12051025498658419, "clip_ratio/region_mean": 0.14134358894079924, "reward_total_mean": 0.6913444399833679, "reward_meter_mean": 0.8658849000930786, "reward_meter_std": 0.2432955801486969, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.45125001668930054, "reward_judge_quality_std": 0.23381541669368744, "reward_total_composite_mean": 0.6913444399833679, "reward_total_composite_std": 0.30673715472221375} {"timestamp_utc": "2026-04-12T23:48:45Z", "mode": "train", "global_step": 662, "epoch": 0.06649924660974385, "loss": -0.0073, "grad_norm": 12.141761779785156, "learning_rate": 7.996969696969697e-06, "num_tokens": 1227743.0, "completions/mean_length": 60.125, "completions/min_length": 54.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.125, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.6988011598587036, "rewards/meter/std": 0.28556013107299805, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9855564832687378, "rewards/repeat_soft/std": 0.015293699689209461, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.12906256318092346, "rewards/total_composite/mean": 0.6575161218643188, "rewards/total_composite/std": 0.14032822847366333, "reward": 0.6575161218643188, "reward_std": 0.14032822847366333, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18923036754131317, "sampling/sampling_logp_difference/max": 1.7330245971679688, "sampling/importance_sampling_ratio/min": 0.17674900591373444, "sampling/importance_sampling_ratio/mean": 1.060889482498169, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9781902134418488, "clip_ratio/low_mean": 0.09385509230196476, "clip_ratio/low_min": 0.09385509230196476, "clip_ratio/high_mean": 0.07823696732521057, "clip_ratio/high_max": 0.07823696732521057, "clip_ratio/region_mean": 0.17209205962717533, "reward_total_mean": 0.6575161218643188, "reward_meter_mean": 0.6988011598587036, "reward_meter_std": 0.28556013107299805, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9855564832687378, "reward_repeat_soft_std": 0.015293699689209461, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.12906256318092346, "reward_total_composite_mean": 0.6575161218643188, "reward_total_composite_std": 0.14032822847366333} {"timestamp_utc": "2026-04-12T23:48:58Z", "mode": "train", "global_step": 663, "epoch": 0.06659969864389755, "loss": -0.215, "grad_norm": 2.3720219135284424, "learning_rate": 7.993939393939396e-06, "num_tokens": 1230453.0, "completions/mean_length": 207.75, "completions/min_length": 95.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 106.33333587646484, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.9211261868476868, "rewards/meter/std": 0.08243218064308167, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.2618614733219147, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.986905038356781, "rewards/repeat_soft/std": 0.025407202541828156, "rewards/judge_quality/mean": 0.1587499976158142, "rewards/judge_quality/std": 0.06577830761671066, "rewards/total_composite/mean": 0.5317012071609497, "rewards/total_composite/std": 0.32994765043258667, "reward": 0.5317012071609497, "reward_std": 0.32994765043258667, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21832983195781708, "sampling/sampling_logp_difference/max": 1.5023975372314453, "sampling/importance_sampling_ratio/min": 0.22259585559368134, "sampling/importance_sampling_ratio/mean": 1.058476448059082, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.393794357776642, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.1471271626651287, "clip_ratio/high_max": 0.1471271626651287, "clip_ratio/region_mean": 0.1471271626651287, "reward_total_mean": 0.5317012071609497, "reward_meter_mean": 0.9211261868476868, "reward_meter_std": 0.08243218064308167, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.2618614733219147, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.986905038356781, "reward_repeat_soft_std": 0.025407202541828156, "reward_judge_quality_mean": 0.1587499976158142, "reward_judge_quality_std": 0.06577830761671066, "reward_total_composite_mean": 0.5317012071609497, "reward_total_composite_std": 0.32994765043258667} {"timestamp_utc": "2026-04-12T23:49:04Z", "mode": "train", "global_step": 664, "epoch": 0.06670015067805123, "loss": 0.1643, "grad_norm": 15.240668296813965, "learning_rate": 7.990909090909091e-06, "num_tokens": 1232098.0, "completions/mean_length": 39.625, "completions/min_length": 32.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.625, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.7249957323074341, "rewards/meter/std": 0.4066072106361389, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9665109515190125, "rewards/repeat_soft/std": 0.042620670050382614, "rewards/judge_quality/mean": 0.7737500071525574, "rewards/judge_quality/std": 0.2260807603597641, "rewards/total_composite/mean": 0.7956491708755493, "rewards/total_composite/std": 0.2684069275856018, "reward": 0.7956491708755493, "reward_std": 0.2684069275856018, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17075417935848236, "sampling/sampling_logp_difference/max": 1.7162823677062988, "sampling/importance_sampling_ratio/min": 0.2672367990016937, "sampling/importance_sampling_ratio/mean": 1.036607265472412, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5069317519664764, "clip_ratio/low_mean": 0.05208333395421505, "clip_ratio/low_min": 0.05208333395421505, "clip_ratio/high_mean": 0.10701710637658834, "clip_ratio/high_max": 0.10701710637658834, "clip_ratio/region_mean": 0.1591004403308034, "reward_total_mean": 0.7956491708755493, "reward_meter_mean": 0.7249957323074341, "reward_meter_std": 0.4066072106361389, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9665109515190125, "reward_repeat_soft_std": 0.042620670050382614, "reward_judge_quality_mean": 0.7737500071525574, "reward_judge_quality_std": 0.2260807603597641, "reward_total_composite_mean": 0.7956491708755493, "reward_total_composite_std": 0.2684069275856018} {"timestamp_utc": "2026-04-12T23:49:11Z", "mode": "train", "global_step": 665, "epoch": 0.06680060271220492, "loss": 0.0291, "grad_norm": 19.42249298095703, "learning_rate": 7.987878787878789e-06, "num_tokens": 1233693.0, "completions/mean_length": 40.375, "completions/min_length": 33.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.375, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.6937024593353271, "rewards/meter/std": 0.4329942464828491, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9948363900184631, "rewards/repeat_soft/std": 0.008596595376729965, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.7438997626304626, "rewards/total_composite/std": 0.183969184756279, "reward": 0.7438997626304626, "reward_std": 0.1839691549539566, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1986127346754074, "sampling/sampling_logp_difference/max": 1.3856356143951416, "sampling/importance_sampling_ratio/min": 0.2501647472381592, "sampling/importance_sampling_ratio/mean": 0.9966184496879578, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9428150057792664, "clip_ratio/low_mean": 0.04466051235795021, "clip_ratio/low_min": 0.04466051235795021, "clip_ratio/high_mean": 0.1624490935355425, "clip_ratio/high_max": 0.1624490935355425, "clip_ratio/region_mean": 0.2071096058934927, "reward_total_mean": 0.7438997626304626, "reward_meter_mean": 0.6937024593353271, "reward_meter_std": 0.4329942464828491, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9948363900184631, "reward_repeat_soft_std": 0.008596595376729965, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.7438997626304626, "reward_total_composite_std": 0.183969184756279} {"timestamp_utc": "2026-04-12T23:49:20Z", "mode": "train", "global_step": 666, "epoch": 0.06690105474635862, "loss": 0.0165, "grad_norm": 15.226571083068848, "learning_rate": 7.984848484848486e-06, "num_tokens": 1235177.0, "completions/mean_length": 30.5, "completions/min_length": 22.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.5, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.5734766125679016, "rewards/meter/std": 0.45357707142829895, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.99312424659729, "rewards/repeat_soft/std": 0.010621348395943642, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.7271268963813782, "rewards/total_composite/std": 0.22504861652851105, "reward": 0.7271268963813782, "reward_std": 0.22504861652851105, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16831670701503754, "sampling/sampling_logp_difference/max": 1.0245132446289062, "sampling/importance_sampling_ratio/min": 0.35897114872932434, "sampling/importance_sampling_ratio/mean": 1.0209767818450928, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6272317469120026, "clip_ratio/low_mean": 0.07877941988408566, "clip_ratio/low_min": 0.07877941988408566, "clip_ratio/high_mean": 0.09001765958964825, "clip_ratio/high_max": 0.09001765958964825, "clip_ratio/region_mean": 0.1687970794737339, "reward_total_mean": 0.7271268963813782, "reward_meter_mean": 0.5734766125679016, "reward_meter_std": 0.45357707142829895, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.99312424659729, "reward_repeat_soft_std": 0.010621348395943642, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.7271268963813782, "reward_total_composite_std": 0.22504861652851105} {"timestamp_utc": "2026-04-12T23:49:26Z", "mode": "train", "global_step": 667, "epoch": 0.06700150678051231, "loss": 0.2164, "grad_norm": 33.85074234008789, "learning_rate": 7.981818181818183e-06, "num_tokens": 1236518.0, "completions/mean_length": 18.625, "completions/min_length": 15.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 18.625, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.8401188850402832, "rewards/meter/std": 0.17421571910381317, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9632102251052856, "rewards/repeat_soft/std": 0.01857268065214157, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7327494621276855, "rewards/total_composite/std": 0.11613330990076065, "reward": 0.7327494621276855, "reward_std": 0.11613330990076065, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16189345717430115, "sampling/sampling_logp_difference/max": 1.2946586608886719, "sampling/importance_sampling_ratio/min": 0.27399134635925293, "sampling/importance_sampling_ratio/mean": 1.029244065284729, "sampling/importance_sampling_ratio/max": 1.9419174194335938, "entropy": 1.8251583725214005, "clip_ratio/low_mean": 0.02557471301406622, "clip_ratio/low_min": 0.02557471301406622, "clip_ratio/high_mean": 0.11423611268401146, "clip_ratio/high_max": 0.11423611268401146, "clip_ratio/region_mean": 0.13981082569807768, "reward_total_mean": 0.7327494621276855, "reward_meter_mean": 0.8401188850402832, "reward_meter_std": 0.17421571910381317, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9632102251052856, "reward_repeat_soft_std": 0.01857268065214157, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7327494621276855, "reward_total_composite_std": 0.11613330990076065} {"timestamp_utc": "2026-04-12T23:49:32Z", "mode": "train", "global_step": 668, "epoch": 0.06710195881466599, "loss": 0.0389, "grad_norm": 22.334552764892578, "learning_rate": 7.978787878787879e-06, "num_tokens": 1238016.0, "completions/mean_length": 27.25, "completions/min_length": 23.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.25, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.41396182775497437, "rewards/meter/std": 0.3503010869026184, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9925290942192078, "rewards/repeat_soft/std": 0.015086323022842407, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.5990357398986816, "rewards/total_composite/std": 0.15343156456947327, "reward": 0.5990357398986816, "reward_std": 0.15343156456947327, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19444991648197174, "sampling/sampling_logp_difference/max": 1.6374788284301758, "sampling/importance_sampling_ratio/min": 0.19446972012519836, "sampling/importance_sampling_ratio/mean": 1.0293306112289429, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2774742543697357, "clip_ratio/low_mean": 0.07095490768551826, "clip_ratio/low_min": 0.07095490768551826, "clip_ratio/high_mean": 0.11864440236240625, "clip_ratio/high_max": 0.11864440236240625, "clip_ratio/region_mean": 0.18959931004792452, "reward_total_mean": 0.5990357398986816, "reward_meter_mean": 0.41396182775497437, "reward_meter_std": 0.3503010869026184, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9925290942192078, "reward_repeat_soft_std": 0.015086323022842407, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.5990357398986816, "reward_total_composite_std": 0.15343156456947327} {"timestamp_utc": "2026-04-12T23:49:39Z", "mode": "train", "global_step": 669, "epoch": 0.06720241084881969, "loss": 0.0248, "grad_norm": 19.409997940063477, "learning_rate": 7.975757575757576e-06, "num_tokens": 1239540.0, "completions/mean_length": 31.5, "completions/min_length": 29.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.5, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.35983675718307495, "rewards/meter/std": 0.39586254954338074, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9485769867897034, "rewards/repeat_soft/std": 0.08132870495319366, "rewards/judge_quality/mean": 0.4024999737739563, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.5275342464447021, "rewards/total_composite/std": 0.18616198003292084, "reward": 0.5275342464447021, "reward_std": 0.18616199493408203, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1752985119819641, "sampling/sampling_logp_difference/max": 1.966348648071289, "sampling/importance_sampling_ratio/min": 0.13996699452400208, "sampling/importance_sampling_ratio/mean": 0.9995108246803284, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.060364991426468, "clip_ratio/low_mean": 0.08407707931473851, "clip_ratio/low_min": 0.08407707931473851, "clip_ratio/high_mean": 0.062316715717315674, "clip_ratio/high_max": 0.062316715717315674, "clip_ratio/region_mean": 0.14639379503205419, "reward_total_mean": 0.5275342464447021, "reward_meter_mean": 0.35983675718307495, "reward_meter_std": 0.39586254954338074, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9485769867897034, "reward_repeat_soft_std": 0.08132870495319366, "reward_judge_quality_mean": 0.4024999737739563, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.5275342464447021, "reward_total_composite_std": 0.18616198003292084} {"timestamp_utc": "2026-04-12T23:49:47Z", "mode": "train", "global_step": 670, "epoch": 0.06730286288297338, "loss": -0.0114, "grad_norm": 6.77335262298584, "learning_rate": 7.972727272727273e-06, "num_tokens": 1242016.0, "completions/mean_length": 129.5, "completions/min_length": 118.0, "completions/max_length": 144.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 129.5, "completions/min_terminated_length": 118.0, "completions/max_terminated_length": 144.0, "rewards/meter/mean": 0.9743355512619019, "rewards/meter/std": 0.03375514969229698, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9542875289916992, "rewards/repeat_soft/std": 0.03427547961473465, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.14520922303199768, "rewards/total_composite/mean": 0.8083797693252563, "rewards/total_composite/std": 0.05107976123690605, "reward": 0.8083797693252563, "reward_std": 0.05107977241277695, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1982131153345108, "sampling/sampling_logp_difference/max": 1.6780953407287598, "sampling/importance_sampling_ratio/min": 0.186729297041893, "sampling/importance_sampling_ratio/mean": 1.0429188013076782, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0594787299633026, "clip_ratio/low_mean": 0.05291538778692484, "clip_ratio/low_min": 0.05291538778692484, "clip_ratio/high_mean": 0.10872476920485497, "clip_ratio/high_max": 0.10872476920485497, "clip_ratio/region_mean": 0.1616401569917798, "reward_total_mean": 0.8083797693252563, "reward_meter_mean": 0.9743355512619019, "reward_meter_std": 0.03375514969229698, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9542875289916992, "reward_repeat_soft_std": 0.03427547961473465, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.14520922303199768, "reward_total_composite_mean": 0.8083797693252563, "reward_total_composite_std": 0.05107976123690605} {"timestamp_utc": "2026-04-12T23:49:54Z", "mode": "train", "global_step": 671, "epoch": 0.06740331491712707, "loss": 0.0576, "grad_norm": 15.472140312194824, "learning_rate": 7.96969696969697e-06, "num_tokens": 1243620.0, "completions/mean_length": 41.5, "completions/min_length": 33.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.5, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.8314551115036011, "rewards/meter/std": 0.23421455919742584, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9842885732650757, "rewards/repeat_soft/std": 0.019820185378193855, "rewards/judge_quality/mean": 0.6312500238418579, "rewards/judge_quality/std": 0.316608726978302, "rewards/total_composite/mean": 0.8119586706161499, "rewards/total_composite/std": 0.1870156079530716, "reward": 0.8119586706161499, "reward_std": 0.1870155781507492, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18926681578159332, "sampling/sampling_logp_difference/max": 1.3547677993774414, "sampling/importance_sampling_ratio/min": 0.2580071985721588, "sampling/importance_sampling_ratio/mean": 1.0557665824890137, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2753549367189407, "clip_ratio/low_mean": 0.04854052234441042, "clip_ratio/low_min": 0.04854052234441042, "clip_ratio/high_mean": 0.09880457818508148, "clip_ratio/high_max": 0.09880457818508148, "clip_ratio/region_mean": 0.1473451005294919, "reward_total_mean": 0.8119586706161499, "reward_meter_mean": 0.8314551115036011, "reward_meter_std": 0.23421455919742584, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9842885732650757, "reward_repeat_soft_std": 0.019820185378193855, "reward_judge_quality_mean": 0.6312500238418579, "reward_judge_quality_std": 0.316608726978302, "reward_total_composite_mean": 0.8119586706161499, "reward_total_composite_std": 0.1870156079530716} {"timestamp_utc": "2026-04-12T23:50:02Z", "mode": "train", "global_step": 672, "epoch": 0.06750376695128077, "loss": 0.0281, "grad_norm": 12.642541885375977, "learning_rate": 7.966666666666668e-06, "num_tokens": 1245508.0, "completions/mean_length": 68.0, "completions/min_length": 59.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.7896327972412109, "rewards/meter/std": 0.3099375069141388, "rewards/count_adherence/mean": 0.7749999761581421, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9893913269042969, "rewards/repeat_soft/std": 0.014315756037831306, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.6901488900184631, "rewards/total_composite/std": 0.14337681233882904, "reward": 0.6901488900184631, "reward_std": 0.14337681233882904, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19489407539367676, "sampling/sampling_logp_difference/max": 1.6700506210327148, "sampling/importance_sampling_ratio/min": 0.18823754787445068, "sampling/importance_sampling_ratio/mean": 1.0411486625671387, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3250881284475327, "clip_ratio/low_mean": 0.08847402967512608, "clip_ratio/low_min": 0.08847402967512608, "clip_ratio/high_mean": 0.09961682558059692, "clip_ratio/high_max": 0.09961682558059692, "clip_ratio/region_mean": 0.188090855255723, "reward_total_mean": 0.6901488900184631, "reward_meter_mean": 0.7896327972412109, "reward_meter_std": 0.3099375069141388, "reward_count_adherence_mean": 0.7749999761581421, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9893913269042969, "reward_repeat_soft_std": 0.014315756037831306, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.6901488900184631, "reward_total_composite_std": 0.14337681233882904} {"timestamp_utc": "2026-04-12T23:50:10Z", "mode": "train", "global_step": 673, "epoch": 0.06760421898543445, "loss": 0.0423, "grad_norm": 13.482980728149414, "learning_rate": 7.963636363636365e-06, "num_tokens": 1247237.0, "completions/mean_length": 56.125, "completions/min_length": 48.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.125, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9557358026504517, "rewards/meter/std": 0.07887719571590424, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9874750375747681, "rewards/repeat_soft/std": 0.013064987026154995, "rewards/judge_quality/mean": 0.41874998807907104, "rewards/judge_quality/std": 0.14574319124221802, "rewards/total_composite/mean": 0.7669535875320435, "rewards/total_composite/std": 0.054405469447374344, "reward": 0.7669535875320435, "reward_std": 0.054405469447374344, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17193223536014557, "sampling/sampling_logp_difference/max": 0.9614276885986328, "sampling/importance_sampling_ratio/min": 0.38234663009643555, "sampling/importance_sampling_ratio/mean": 1.0282665491104126, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0395725667476654, "clip_ratio/low_mean": 0.046586678363382816, "clip_ratio/low_min": 0.046586678363382816, "clip_ratio/high_mean": 0.11617154814302921, "clip_ratio/high_max": 0.11617154814302921, "clip_ratio/region_mean": 0.16275822650641203, "reward_total_mean": 0.7669535875320435, "reward_meter_mean": 0.9557358026504517, "reward_meter_std": 0.07887719571590424, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9874750375747681, "reward_repeat_soft_std": 0.013064987026154995, "reward_judge_quality_mean": 0.41874998807907104, "reward_judge_quality_std": 0.14574319124221802, "reward_total_composite_mean": 0.7669535875320435, "reward_total_composite_std": 0.054405469447374344} {"timestamp_utc": "2026-04-12T23:50:18Z", "mode": "train", "global_step": 674, "epoch": 0.06770467101958814, "loss": 0.0922, "grad_norm": 10.71855354309082, "learning_rate": 7.96060606060606e-06, "num_tokens": 1249572.0, "completions/mean_length": 104.875, "completions/min_length": 86.0, "completions/max_length": 127.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.875, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.9093436002731323, "rewards/meter/std": 0.11297979205846786, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.08625821024179459, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9627379775047302, "rewards/repeat_soft/std": 0.032340697944164276, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.6792948246002197, "rewards/total_composite/std": 0.28118887543678284, "reward": 0.6792948246002197, "reward_std": 0.28118887543678284, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1850101202726364, "sampling/sampling_logp_difference/max": 1.740124225616455, "sampling/importance_sampling_ratio/min": 0.1754986047744751, "sampling/importance_sampling_ratio/mean": 1.028924584388733, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6607487946748734, "clip_ratio/low_mean": 0.03293010778725147, "clip_ratio/low_min": 0.03293010778725147, "clip_ratio/high_mean": 0.1275601638481021, "clip_ratio/high_max": 0.1275601638481021, "clip_ratio/region_mean": 0.16049027163535357, "reward_total_mean": 0.6792948246002197, "reward_meter_mean": 0.9093436002731323, "reward_meter_std": 0.11297979205846786, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.08625821024179459, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9627379775047302, "reward_repeat_soft_std": 0.032340697944164276, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.6792948246002197, "reward_total_composite_std": 0.28118887543678284} {"timestamp_utc": "2026-04-12T23:50:24Z", "mode": "train", "global_step": 675, "epoch": 0.06780512305374184, "loss": 0.068, "grad_norm": 13.913802146911621, "learning_rate": 7.957575757575758e-06, "num_tokens": 1251258.0, "completions/mean_length": 41.75, "completions/min_length": 37.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.75, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.736784815788269, "rewards/meter/std": 0.15707732737064362, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9957863092422485, "rewards/repeat_soft/std": 0.005170407239347696, "rewards/judge_quality/mean": 0.44749999046325684, "rewards/judge_quality/std": 0.20824094116687775, "rewards/total_composite/mean": 0.7153818011283875, "rewards/total_composite/std": 0.10043212026357651, "reward": 0.7153818011283875, "reward_std": 0.1004321277141571, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1910848617553711, "sampling/sampling_logp_difference/max": 1.3257203102111816, "sampling/importance_sampling_ratio/min": 0.34032779932022095, "sampling/importance_sampling_ratio/mean": 1.0591437816619873, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0734806656837463, "clip_ratio/low_mean": 0.09793432429432869, "clip_ratio/low_min": 0.09793432429432869, "clip_ratio/high_mean": 0.08077990636229515, "clip_ratio/high_max": 0.08077990636229515, "clip_ratio/region_mean": 0.17871423065662384, "reward_total_mean": 0.7153818011283875, "reward_meter_mean": 0.736784815788269, "reward_meter_std": 0.15707732737064362, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9957863092422485, "reward_repeat_soft_std": 0.005170407239347696, "reward_judge_quality_mean": 0.44749999046325684, "reward_judge_quality_std": 0.20824094116687775, "reward_total_composite_mean": 0.7153818011283875, "reward_total_composite_std": 0.10043212026357651} {"timestamp_utc": "2026-04-12T23:50:31Z", "mode": "train", "global_step": 676, "epoch": 0.06790557508789553, "loss": 0.1159, "grad_norm": 10.174636840820312, "learning_rate": 7.954545454545455e-06, "num_tokens": 1253349.0, "completions/mean_length": 89.375, "completions/min_length": 70.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.375, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.7026269435882568, "rewards/meter/std": 0.2773953974246979, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9937667846679688, "rewards/repeat_soft/std": 0.008926589041948318, "rewards/judge_quality/mean": 0.59375, "rewards/judge_quality/std": 0.19234922528266907, "rewards/total_composite/mean": 0.7324337959289551, "rewards/total_composite/std": 0.16054359078407288, "reward": 0.7324337959289551, "reward_std": 0.16054359078407288, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1751483976840973, "sampling/sampling_logp_difference/max": 1.3513903617858887, "sampling/importance_sampling_ratio/min": 0.258880078792572, "sampling/importance_sampling_ratio/mean": 1.050718069076538, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5733711272478104, "clip_ratio/low_mean": 0.034381807781755924, "clip_ratio/low_min": 0.034381807781755924, "clip_ratio/high_mean": 0.10083239804953337, "clip_ratio/high_max": 0.10083239804953337, "clip_ratio/region_mean": 0.1352142058312893, "reward_total_mean": 0.7324337959289551, "reward_meter_mean": 0.7026269435882568, "reward_meter_std": 0.2773953974246979, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9937667846679688, "reward_repeat_soft_std": 0.008926589041948318, "reward_judge_quality_mean": 0.59375, "reward_judge_quality_std": 0.19234922528266907, "reward_total_composite_mean": 0.7324337959289551, "reward_total_composite_std": 0.16054359078407288} {"timestamp_utc": "2026-04-12T23:50:38Z", "mode": "train", "global_step": 677, "epoch": 0.06800602712204921, "loss": 0.055, "grad_norm": 15.376818656921387, "learning_rate": 7.951515151515152e-06, "num_tokens": 1255109.0, "completions/mean_length": 43.0, "completions/min_length": 35.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.6734113097190857, "rewards/meter/std": 0.39189913868904114, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9779559969902039, "rewards/repeat_soft/std": 0.02684602700173855, "rewards/judge_quality/mean": 0.518750011920929, "rewards/judge_quality/std": 0.24746933579444885, "rewards/total_composite/mean": 0.6449941396713257, "rewards/total_composite/std": 0.32417988777160645, "reward": 0.6449941396713257, "reward_std": 0.32417985796928406, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19888074696063995, "sampling/sampling_logp_difference/max": 1.244767189025879, "sampling/importance_sampling_ratio/min": 0.2880079448223114, "sampling/importance_sampling_ratio/mean": 1.041545033454895, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7454169243574142, "clip_ratio/low_mean": 0.04942788928747177, "clip_ratio/low_min": 0.04942788928747177, "clip_ratio/high_mean": 0.11215767078101635, "clip_ratio/high_max": 0.11215767078101635, "clip_ratio/region_mean": 0.16158556006848812, "reward_total_mean": 0.6449941396713257, "reward_meter_mean": 0.6734113097190857, "reward_meter_std": 0.39189913868904114, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9779559969902039, "reward_repeat_soft_std": 0.02684602700173855, "reward_judge_quality_mean": 0.518750011920929, "reward_judge_quality_std": 0.24746933579444885, "reward_total_composite_mean": 0.6449941396713257, "reward_total_composite_std": 0.32417988777160645} {"timestamp_utc": "2026-04-12T23:50:46Z", "mode": "train", "global_step": 678, "epoch": 0.06810647915620291, "loss": 0.0192, "grad_norm": 15.603260040283203, "learning_rate": 7.948484848484848e-06, "num_tokens": 1256660.0, "completions/mean_length": 39.875, "completions/min_length": 32.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.875, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.5449224710464478, "rewards/meter/std": 0.3538571298122406, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9955349564552307, "rewards/repeat_soft/std": 0.0081853736191988, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.24833375215530396, "rewards/total_composite/mean": 0.6541435718536377, "rewards/total_composite/std": 0.20706556737422943, "reward": 0.6541435718536377, "reward_std": 0.20706553757190704, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1865261048078537, "sampling/sampling_logp_difference/max": 1.095881462097168, "sampling/importance_sampling_ratio/min": 0.33424484729766846, "sampling/importance_sampling_ratio/mean": 1.03995943069458, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8737338036298752, "clip_ratio/low_mean": 0.06655425392091274, "clip_ratio/low_min": 0.06655425392091274, "clip_ratio/high_mean": 0.08403563220053911, "clip_ratio/high_max": 0.08403563220053911, "clip_ratio/region_mean": 0.15058988612145185, "reward_total_mean": 0.6541435718536377, "reward_meter_mean": 0.5449224710464478, "reward_meter_std": 0.3538571298122406, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9955349564552307, "reward_repeat_soft_std": 0.0081853736191988, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.24833375215530396, "reward_total_composite_mean": 0.6541435718536377, "reward_total_composite_std": 0.20706556737422943} {"timestamp_utc": "2026-04-12T23:50:58Z", "mode": "train", "global_step": 679, "epoch": 0.0682069311903566, "loss": -0.088, "grad_norm": 2.2537970542907715, "learning_rate": 7.945454545454547e-06, "num_tokens": 1258371.0, "completions/mean_length": 158.875, "completions/min_length": 35.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 41.16666793823242, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.40461593866348267, "rewards/meter/std": 0.35396808385849, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9916712045669556, "rewards/repeat_soft/std": 0.011228786781430244, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.1718803495168686, "rewards/total_composite/mean": 0.4468975067138672, "rewards/total_composite/std": 0.31138914823532104, "reward": 0.4468975067138672, "reward_std": 0.31138914823532104, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22368258237838745, "sampling/sampling_logp_difference/max": 1.3080158233642578, "sampling/importance_sampling_ratio/min": 0.2703559696674347, "sampling/importance_sampling_ratio/mean": 1.0519258975982666, "sampling/importance_sampling_ratio/max": 1.9606502056121826, "entropy": 2.2866038382053375, "clip_ratio/low_mean": 0.039638932794332504, "clip_ratio/low_min": 0.039638932794332504, "clip_ratio/high_mean": 0.10139270685613155, "clip_ratio/high_max": 0.10139270685613155, "clip_ratio/region_mean": 0.14103163965046406, "reward_total_mean": 0.4468975067138672, "reward_meter_mean": 0.40461593866348267, "reward_meter_std": 0.35396808385849, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9916712045669556, "reward_repeat_soft_std": 0.011228786781430244, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.1718803495168686, "reward_total_composite_mean": 0.4468975067138672, "reward_total_composite_std": 0.31138914823532104} {"timestamp_utc": "2026-04-12T23:51:05Z", "mode": "train", "global_step": 680, "epoch": 0.0683073832245103, "loss": 0.0711, "grad_norm": 12.648969650268555, "learning_rate": 7.942424242424242e-06, "num_tokens": 1259893.0, "completions/mean_length": 42.25, "completions/min_length": 34.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.25, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.9938188791275024, "rewards/meter/std": 0.00330791505984962, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9780452847480774, "rewards/repeat_soft/std": 0.019392991438508034, "rewards/judge_quality/mean": 0.5024999976158142, "rewards/judge_quality/std": 0.1348809152841568, "rewards/total_composite/mean": 0.8457729816436768, "rewards/total_composite/std": 0.04079907014966011, "reward": 0.8457729816436768, "reward_std": 0.04079905524849892, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17229965329170227, "sampling/sampling_logp_difference/max": 1.449167251586914, "sampling/importance_sampling_ratio/min": 0.23476570844650269, "sampling/importance_sampling_ratio/mean": 1.0301945209503174, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6530685126781464, "clip_ratio/low_mean": 0.11077918205410242, "clip_ratio/low_min": 0.11077918205410242, "clip_ratio/high_mean": 0.03973214328289032, "clip_ratio/high_max": 0.03973214328289032, "clip_ratio/region_mean": 0.15051132533699274, "reward_total_mean": 0.8457729816436768, "reward_meter_mean": 0.9938188791275024, "reward_meter_std": 0.00330791505984962, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9780452847480774, "reward_repeat_soft_std": 0.019392991438508034, "reward_judge_quality_mean": 0.5024999976158142, "reward_judge_quality_std": 0.1348809152841568, "reward_total_composite_mean": 0.8457729816436768, "reward_total_composite_std": 0.04079907014966011} {"timestamp_utc": "2026-04-12T23:51:18Z", "mode": "train", "global_step": 681, "epoch": 0.068407835258664, "loss": -0.1241, "grad_norm": 2.22104549407959, "learning_rate": 7.93939393939394e-06, "num_tokens": 1261615.0, "completions/mean_length": 166.25, "completions/min_length": 45.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 51.0, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.6998183727264404, "rewards/meter/std": 0.33605706691741943, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9851112365722656, "rewards/repeat_soft/std": 0.03240746632218361, "rewards/judge_quality/mean": 0.3137499988079071, "rewards/judge_quality/std": 0.2838982343673706, "rewards/total_composite/mean": 0.5423867106437683, "rewards/total_composite/std": 0.3462289571762085, "reward": 0.5423867106437683, "reward_std": 0.3462289571762085, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21077917516231537, "sampling/sampling_logp_difference/max": 1.3661327362060547, "sampling/importance_sampling_ratio/min": 0.2550915777683258, "sampling/importance_sampling_ratio/mean": 1.0287302732467651, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6018331795930862, "clip_ratio/low_mean": 0.01822916604578495, "clip_ratio/low_min": 0.01822916604578495, "clip_ratio/high_mean": 0.10775362513959408, "clip_ratio/high_max": 0.10775362513959408, "clip_ratio/region_mean": 0.12598279118537903, "reward_total_mean": 0.5423867106437683, "reward_meter_mean": 0.6998183727264404, "reward_meter_std": 0.33605706691741943, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9851112365722656, "reward_repeat_soft_std": 0.03240746632218361, "reward_judge_quality_mean": 0.3137499988079071, "reward_judge_quality_std": 0.2838982343673706, "reward_total_composite_mean": 0.5423867106437683, "reward_total_composite_std": 0.3462289571762085} {"timestamp_utc": "2026-04-12T23:51:25Z", "mode": "train", "global_step": 682, "epoch": 0.06850828729281767, "loss": 0.0712, "grad_norm": 16.03302574157715, "learning_rate": 7.936363636363637e-06, "num_tokens": 1263372.0, "completions/mean_length": 44.625, "completions/min_length": 32.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.625, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.6484029293060303, "rewards/meter/std": 0.2905818521976471, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9874111413955688, "rewards/repeat_soft/std": 0.011620234698057175, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.6561473608016968, "rewards/total_composite/std": 0.1263122260570526, "reward": 0.6561473608016968, "reward_std": 0.12631221115589142, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19451336562633514, "sampling/sampling_logp_difference/max": 1.663179874420166, "sampling/importance_sampling_ratio/min": 0.18953531980514526, "sampling/importance_sampling_ratio/mean": 1.0346096754074097, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3661597520112991, "clip_ratio/low_mean": 0.055562468245625496, "clip_ratio/low_min": 0.055562468245625496, "clip_ratio/high_mean": 0.11883308365941048, "clip_ratio/high_max": 0.11883308365941048, "clip_ratio/region_mean": 0.17439555190503597, "reward_total_mean": 0.6561473608016968, "reward_meter_mean": 0.6484029293060303, "reward_meter_std": 0.2905818521976471, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9874111413955688, "reward_repeat_soft_std": 0.011620234698057175, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.6561473608016968, "reward_total_composite_std": 0.1263122260570526} {"timestamp_utc": "2026-04-12T23:51:32Z", "mode": "train", "global_step": 683, "epoch": 0.06860873932697137, "loss": -0.035, "grad_norm": 12.34382152557373, "learning_rate": 7.933333333333334e-06, "num_tokens": 1265072.0, "completions/mean_length": 52.5, "completions/min_length": 45.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.5, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9880703687667847, "rewards/meter/std": 0.009158551692962646, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9796148538589478, "rewards/repeat_soft/std": 0.01974349096417427, "rewards/judge_quality/mean": 0.304999977350235, "rewards/judge_quality/std": 0.10928337275981903, "rewards/total_composite/mean": 0.7840931415557861, "rewards/total_composite/std": 0.034285061061382294, "reward": 0.7840931415557861, "reward_std": 0.03428506478667259, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19868707656860352, "sampling/sampling_logp_difference/max": 1.387216567993164, "sampling/importance_sampling_ratio/min": 0.24976955354213715, "sampling/importance_sampling_ratio/mean": 1.046642541885376, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3714460730552673, "clip_ratio/low_mean": 0.08516339212656021, "clip_ratio/low_min": 0.08516339212656021, "clip_ratio/high_mean": 0.05503600463271141, "clip_ratio/high_max": 0.05503600463271141, "clip_ratio/region_mean": 0.14019939675927162, "reward_total_mean": 0.7840931415557861, "reward_meter_mean": 0.9880703687667847, "reward_meter_std": 0.009158551692962646, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9796148538589478, "reward_repeat_soft_std": 0.01974349096417427, "reward_judge_quality_mean": 0.304999977350235, "reward_judge_quality_std": 0.10928337275981903, "reward_total_composite_mean": 0.7840931415557861, "reward_total_composite_std": 0.034285061061382294} {"timestamp_utc": "2026-04-12T23:51:40Z", "mode": "train", "global_step": 684, "epoch": 0.06870919136112506, "loss": 0.0033, "grad_norm": 13.900835990905762, "learning_rate": 7.930303030303031e-06, "num_tokens": 1266531.0, "completions/mean_length": 38.375, "completions/min_length": 30.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.375, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.7319329977035522, "rewards/meter/std": 0.38230088353157043, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9662365913391113, "rewards/repeat_soft/std": 0.02535836584866047, "rewards/judge_quality/mean": 0.5149999856948853, "rewards/judge_quality/std": 0.2676885426044464, "rewards/total_composite/mean": 0.7304935455322266, "rewards/total_composite/std": 0.20268788933753967, "reward": 0.7304935455322266, "reward_std": 0.20268788933753967, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1671430766582489, "sampling/sampling_logp_difference/max": 1.3728899955749512, "sampling/importance_sampling_ratio/min": 0.2533736526966095, "sampling/importance_sampling_ratio/mean": 1.034939169883728, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3912553563714027, "clip_ratio/low_mean": 0.04298780392855406, "clip_ratio/low_min": 0.04298780392855406, "clip_ratio/high_mean": 0.10993447713553905, "clip_ratio/high_max": 0.10993447713553905, "clip_ratio/region_mean": 0.1529222810640931, "reward_total_mean": 0.7304935455322266, "reward_meter_mean": 0.7319329977035522, "reward_meter_std": 0.38230088353157043, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9662365913391113, "reward_repeat_soft_std": 0.02535836584866047, "reward_judge_quality_mean": 0.5149999856948853, "reward_judge_quality_std": 0.2676885426044464, "reward_total_composite_mean": 0.7304935455322266, "reward_total_composite_std": 0.20268788933753967} {"timestamp_utc": "2026-04-12T23:51:53Z", "mode": "train", "global_step": 685, "epoch": 0.06880964339527876, "loss": -0.1266, "grad_norm": 2.2217793464660645, "learning_rate": 7.927272727272729e-06, "num_tokens": 1268142.0, "completions/mean_length": 100.375, "completions/min_length": 38.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 41.57143020629883, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.8188420534133911, "rewards/meter/std": 0.3327735364437103, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9962054491043091, "rewards/repeat_soft/std": 0.004047216381877661, "rewards/judge_quality/mean": 0.42250001430511475, "rewards/judge_quality/std": 0.1880539357662201, "rewards/total_composite/mean": 0.709684431552887, "rewards/total_composite/std": 0.29359325766563416, "reward": 0.709684431552887, "reward_std": 0.29359325766563416, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17032833397388458, "sampling/sampling_logp_difference/max": 0.9885916709899902, "sampling/importance_sampling_ratio/min": 0.3721003830432892, "sampling/importance_sampling_ratio/mean": 1.0369651317596436, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6862024366855621, "clip_ratio/low_mean": 0.016447369009256363, "clip_ratio/low_min": 0.016447369009256363, "clip_ratio/high_mean": 0.11805519182235003, "clip_ratio/high_max": 0.11805519182235003, "clip_ratio/region_mean": 0.1345025608316064, "reward_total_mean": 0.709684431552887, "reward_meter_mean": 0.8188420534133911, "reward_meter_std": 0.3327735364437103, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9962054491043091, "reward_repeat_soft_std": 0.004047216381877661, "reward_judge_quality_mean": 0.42250001430511475, "reward_judge_quality_std": 0.1880539357662201, "reward_total_composite_mean": 0.709684431552887, "reward_total_composite_std": 0.29359325766563416} {"timestamp_utc": "2026-04-12T23:51:59Z", "mode": "train", "global_step": 686, "epoch": 0.06891009542943245, "loss": 0.0817, "grad_norm": 16.122390747070312, "learning_rate": 7.924242424242426e-06, "num_tokens": 1269731.0, "completions/mean_length": 40.625, "completions/min_length": 29.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.625, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.339469850063324, "rewards/meter/std": 0.4074195623397827, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9784785509109497, "rewards/repeat_soft/std": 0.01973510906100273, "rewards/judge_quality/mean": 0.39249998331069946, "rewards/judge_quality/std": 0.08892211318016052, "rewards/total_composite/mean": 0.4162948727607727, "rewards/total_composite/std": 0.22644895315170288, "reward": 0.4162948727607727, "reward_std": 0.22644895315170288, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20657594501972198, "sampling/sampling_logp_difference/max": 1.1736583709716797, "sampling/importance_sampling_ratio/min": 0.3092336058616638, "sampling/importance_sampling_ratio/mean": 1.045551061630249, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.041818156838417, "clip_ratio/low_mean": 0.07907605171203613, "clip_ratio/low_min": 0.07907605171203613, "clip_ratio/high_mean": 0.07594279013574123, "clip_ratio/high_max": 0.07594279013574123, "clip_ratio/region_mean": 0.15501884184777737, "reward_total_mean": 0.4162948727607727, "reward_meter_mean": 0.339469850063324, "reward_meter_std": 0.4074195623397827, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9784785509109497, "reward_repeat_soft_std": 0.01973510906100273, "reward_judge_quality_mean": 0.39249998331069946, "reward_judge_quality_std": 0.08892211318016052, "reward_total_composite_mean": 0.4162948727607727, "reward_total_composite_std": 0.22644895315170288} {"timestamp_utc": "2026-04-12T23:52:06Z", "mode": "train", "global_step": 687, "epoch": 0.06901054746358613, "loss": -0.0437, "grad_norm": 15.222193717956543, "learning_rate": 7.921212121212122e-06, "num_tokens": 1271239.0, "completions/mean_length": 35.5, "completions/min_length": 29.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.5, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.6800107955932617, "rewards/meter/std": 0.3714495003223419, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9948931932449341, "rewards/repeat_soft/std": 0.0058179874904453754, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.25150617957115173, "rewards/total_composite/mean": 0.7126191854476929, "rewards/total_composite/std": 0.20966067910194397, "reward": 0.7126191854476929, "reward_std": 0.20966067910194397, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18004098534584045, "sampling/sampling_logp_difference/max": 2.2782444953918457, "sampling/importance_sampling_ratio/min": 0.10246392339468002, "sampling/importance_sampling_ratio/mean": 1.0178937911987305, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4125706851482391, "clip_ratio/low_mean": 0.0936942994594574, "clip_ratio/low_min": 0.0936942994594574, "clip_ratio/high_mean": 0.11315207928419113, "clip_ratio/high_max": 0.11315207928419113, "clip_ratio/region_mean": 0.20684637874364853, "reward_total_mean": 0.7126191854476929, "reward_meter_mean": 0.6800107955932617, "reward_meter_std": 0.3714495003223419, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9948931932449341, "reward_repeat_soft_std": 0.0058179874904453754, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.25150617957115173, "reward_total_composite_mean": 0.7126191854476929, "reward_total_composite_std": 0.20966067910194397} {"timestamp_utc": "2026-04-12T23:52:17Z", "mode": "train", "global_step": 688, "epoch": 0.06911099949773983, "loss": -0.1419, "grad_norm": 3.0668346881866455, "learning_rate": 7.918181818181819e-06, "num_tokens": 1272804.0, "completions/mean_length": 105.625, "completions/min_length": 35.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 47.57143020629883, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.870958685874939, "rewards/meter/std": 0.30682173371315, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9754266738891602, "rewards/repeat_soft/std": 0.0168205164372921, "rewards/judge_quality/mean": 0.3499999940395355, "rewards/judge_quality/std": 0.15445756912231445, "rewards/total_composite/mean": 0.6604483127593994, "rewards/total_composite/std": 0.29862484335899353, "reward": 0.6604483127593994, "reward_std": 0.29862481355667114, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17789314687252045, "sampling/sampling_logp_difference/max": 1.0222740173339844, "sampling/importance_sampling_ratio/min": 0.3597759008407593, "sampling/importance_sampling_ratio/mean": 1.0255573987960815, "sampling/importance_sampling_ratio/max": 1.8489559888839722, "entropy": 1.6944590508937836, "clip_ratio/low_mean": 0.02142857201397419, "clip_ratio/low_min": 0.02142857201397419, "clip_ratio/high_mean": 0.1381912538781762, "clip_ratio/high_max": 0.1381912538781762, "clip_ratio/region_mean": 0.1596198258921504, "reward_total_mean": 0.6604483127593994, "reward_meter_mean": 0.870958685874939, "reward_meter_std": 0.30682173371315, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9754266738891602, "reward_repeat_soft_std": 0.0168205164372921, "reward_judge_quality_mean": 0.3499999940395355, "reward_judge_quality_std": 0.15445756912231445, "reward_total_composite_mean": 0.6604483127593994, "reward_total_composite_std": 0.29862484335899353} {"timestamp_utc": "2026-04-12T23:52:25Z", "mode": "train", "global_step": 689, "epoch": 0.06921145153189352, "loss": 0.0428, "grad_norm": 12.836787223815918, "learning_rate": 7.915151515151516e-06, "num_tokens": 1274713.0, "completions/mean_length": 71.625, "completions/min_length": 59.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.625, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.5991175174713135, "rewards/meter/std": 0.3549479842185974, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.989762544631958, "rewards/repeat_soft/std": 0.020978068932890892, "rewards/judge_quality/mean": 0.3687500059604645, "rewards/judge_quality/std": 0.10802611708641052, "rewards/total_composite/mean": 0.6292040944099426, "rewards/total_composite/std": 0.16367308795452118, "reward": 0.6292040944099426, "reward_std": 0.16367308795452118, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21065233647823334, "sampling/sampling_logp_difference/max": 1.7434654235839844, "sampling/importance_sampling_ratio/min": 0.1749131977558136, "sampling/importance_sampling_ratio/mean": 1.0315591096878052, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.175749897956848, "clip_ratio/low_mean": 0.0631527453660965, "clip_ratio/low_min": 0.0631527453660965, "clip_ratio/high_mean": 0.09727823175489902, "clip_ratio/high_max": 0.09727823175489902, "clip_ratio/region_mean": 0.16043097712099552, "reward_total_mean": 0.6292040944099426, "reward_meter_mean": 0.5991175174713135, "reward_meter_std": 0.3549479842185974, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.989762544631958, "reward_repeat_soft_std": 0.020978068932890892, "reward_judge_quality_mean": 0.3687500059604645, "reward_judge_quality_std": 0.10802611708641052, "reward_total_composite_mean": 0.6292040944099426, "reward_total_composite_std": 0.16367308795452118} {"timestamp_utc": "2026-04-12T23:52:32Z", "mode": "train", "global_step": 690, "epoch": 0.06931190356604722, "loss": -0.0557, "grad_norm": 18.964567184448242, "learning_rate": 7.912121212121213e-06, "num_tokens": 1276048.0, "completions/mean_length": 22.875, "completions/min_length": 16.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.875, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.8793078660964966, "rewards/meter/std": 0.10083417594432831, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9621384143829346, "rewards/repeat_soft/std": 0.0010226722806692123, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.8451523780822754, "rewards/total_composite/std": 0.07548144459724426, "reward": 0.8451523780822754, "reward_std": 0.07548145204782486, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18055224418640137, "sampling/sampling_logp_difference/max": 1.3433647155761719, "sampling/importance_sampling_ratio/min": 0.26096612215042114, "sampling/importance_sampling_ratio/mean": 0.9971995949745178, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.804214671254158, "clip_ratio/low_mean": 0.072276103310287, "clip_ratio/low_min": 0.072276103310287, "clip_ratio/high_mean": 0.13333333563059568, "clip_ratio/high_max": 0.13333333563059568, "clip_ratio/region_mean": 0.20560943894088268, "reward_total_mean": 0.8451523780822754, "reward_meter_mean": 0.8793078660964966, "reward_meter_std": 0.10083417594432831, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9621384143829346, "reward_repeat_soft_std": 0.0010226722806692123, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.8451523780822754, "reward_total_composite_std": 0.07548144459724426} {"timestamp_utc": "2026-04-12T23:52:38Z", "mode": "train", "global_step": 691, "epoch": 0.0694123556002009, "loss": 0.0833, "grad_norm": 8.674710273742676, "learning_rate": 7.909090909090909e-06, "num_tokens": 1277492.0, "completions/mean_length": 44.5, "completions/min_length": 33.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.5, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9846251010894775, "rewards/meter/std": 0.008710517548024654, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9523773193359375, "rewards/repeat_soft/std": 0.048489928245544434, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.8375690579414368, "rewards/total_composite/std": 0.05048173666000366, "reward": 0.8375690579414368, "reward_std": 0.050481732934713364, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1547018140554428, "sampling/sampling_logp_difference/max": 1.311014175415039, "sampling/importance_sampling_ratio/min": 0.2695465683937073, "sampling/importance_sampling_ratio/mean": 1.0258877277374268, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5300754308700562, "clip_ratio/low_mean": 0.11067171208560467, "clip_ratio/low_min": 0.11067171208560467, "clip_ratio/high_mean": 0.022058824077248573, "clip_ratio/high_max": 0.022058824077248573, "clip_ratio/region_mean": 0.13273053616285324, "reward_total_mean": 0.8375690579414368, "reward_meter_mean": 0.9846251010894775, "reward_meter_std": 0.008710517548024654, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9523773193359375, "reward_repeat_soft_std": 0.048489928245544434, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.8375690579414368, "reward_total_composite_std": 0.05048173666000366} {"timestamp_utc": "2026-04-12T23:52:45Z", "mode": "train", "global_step": 692, "epoch": 0.06951280763435459, "loss": -0.0415, "grad_norm": 27.81766700744629, "learning_rate": 7.906060606060608e-06, "num_tokens": 1279112.0, "completions/mean_length": 40.5, "completions/min_length": 34.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.5, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.6080271601676941, "rewards/meter/std": 0.3086026608943939, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9696111679077148, "rewards/repeat_soft/std": 0.04605637490749359, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.6664483547210693, "rewards/total_composite/std": 0.1611243486404419, "reward": 0.6664483547210693, "reward_std": 0.1611243039369583, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18400675058364868, "sampling/sampling_logp_difference/max": 1.5996150970458984, "sampling/importance_sampling_ratio/min": 0.20197424292564392, "sampling/importance_sampling_ratio/mean": 1.029883861541748, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5457495748996735, "clip_ratio/low_mean": 0.05444015422835946, "clip_ratio/low_min": 0.05444015422835946, "clip_ratio/high_mean": 0.08449302753433585, "clip_ratio/high_max": 0.08449302753433585, "clip_ratio/region_mean": 0.1389331817626953, "reward_total_mean": 0.6664483547210693, "reward_meter_mean": 0.6080271601676941, "reward_meter_std": 0.3086026608943939, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9696111679077148, "reward_repeat_soft_std": 0.04605637490749359, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.6664483547210693, "reward_total_composite_std": 0.1611243486404419} {"timestamp_utc": "2026-04-12T23:52:53Z", "mode": "train", "global_step": 693, "epoch": 0.06961325966850829, "loss": 0.1455, "grad_norm": 15.290467262268066, "learning_rate": 7.903030303030303e-06, "num_tokens": 1280887.0, "completions/mean_length": 56.875, "completions/min_length": 35.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.875, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.5692048072814941, "rewards/meter/std": 0.30644845962524414, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.17817415297031403, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.994937539100647, "rewards/repeat_soft/std": 0.006808980368077755, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.6037372350692749, "rewards/total_composite/std": 0.2857583165168762, "reward": 0.6037372350692749, "reward_std": 0.2857583165168762, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17668458819389343, "sampling/sampling_logp_difference/max": 1.6584134101867676, "sampling/importance_sampling_ratio/min": 0.19044089317321777, "sampling/importance_sampling_ratio/mean": 1.0297566652297974, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.371073141694069, "clip_ratio/low_mean": 0.06162252277135849, "clip_ratio/low_min": 0.06162252277135849, "clip_ratio/high_mean": 0.10856485180556774, "clip_ratio/high_max": 0.10856485180556774, "clip_ratio/region_mean": 0.17018737457692623, "reward_total_mean": 0.6037372350692749, "reward_meter_mean": 0.5692048072814941, "reward_meter_std": 0.30644845962524414, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.17817415297031403, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.994937539100647, "reward_repeat_soft_std": 0.006808980368077755, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.6037372350692749, "reward_total_composite_std": 0.2857583165168762} {"timestamp_utc": "2026-04-12T23:53:01Z", "mode": "train", "global_step": 694, "epoch": 0.06971371170266198, "loss": 0.1658, "grad_norm": 14.1334228515625, "learning_rate": 7.9e-06, "num_tokens": 1282561.0, "completions/mean_length": 50.25, "completions/min_length": 39.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.25, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.738267183303833, "rewards/meter/std": 0.43586769700050354, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9664202928543091, "rewards/repeat_soft/std": 0.035734958946704865, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.22403763234615326, "rewards/total_composite/mean": 0.7116122245788574, "rewards/total_composite/std": 0.24688629806041718, "reward": 0.7116122245788574, "reward_std": 0.2468862533569336, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18375267088413239, "sampling/sampling_logp_difference/max": 3.9834723472595215, "sampling/importance_sampling_ratio/min": 0.018620869144797325, "sampling/importance_sampling_ratio/mean": 0.9822324514389038, "sampling/importance_sampling_ratio/max": 1.544270634651184, "entropy": 1.239442266523838, "clip_ratio/low_mean": 0.02774390298873186, "clip_ratio/low_min": 0.02774390298873186, "clip_ratio/high_mean": 0.13037077523767948, "clip_ratio/high_max": 0.13037077523767948, "clip_ratio/region_mean": 0.15811467822641134, "reward_total_mean": 0.7116122245788574, "reward_meter_mean": 0.738267183303833, "reward_meter_std": 0.43586769700050354, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9664202928543091, "reward_repeat_soft_std": 0.035734958946704865, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.22403763234615326, "reward_total_composite_mean": 0.7116122245788574, "reward_total_composite_std": 0.24688629806041718} {"timestamp_utc": "2026-04-12T23:53:08Z", "mode": "train", "global_step": 695, "epoch": 0.06981416373681568, "loss": 0.041, "grad_norm": 13.345393180847168, "learning_rate": 7.896969696969698e-06, "num_tokens": 1284340.0, "completions/mean_length": 49.375, "completions/min_length": 34.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.375, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.5419841408729553, "rewards/meter/std": 0.3790515661239624, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9795982837677002, "rewards/repeat_soft/std": 0.033683452755212784, "rewards/judge_quality/mean": 0.8575000166893005, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.7491027116775513, "rewards/total_composite/std": 0.1509392112493515, "reward": 0.7491027116775513, "reward_std": 0.1509392112493515, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17092303931713104, "sampling/sampling_logp_difference/max": 1.5106141567230225, "sampling/importance_sampling_ratio/min": 0.2207743525505066, "sampling/importance_sampling_ratio/mean": 1.017750859260559, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6163691282272339, "clip_ratio/low_mean": 0.07144819293171167, "clip_ratio/low_min": 0.07144819293171167, "clip_ratio/high_mean": 0.08309823088347912, "clip_ratio/high_max": 0.08309823088347912, "clip_ratio/region_mean": 0.1545464238151908, "reward_total_mean": 0.7491027116775513, "reward_meter_mean": 0.5419841408729553, "reward_meter_std": 0.3790515661239624, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9795982837677002, "reward_repeat_soft_std": 0.033683452755212784, "reward_judge_quality_mean": 0.8575000166893005, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.7491027116775513, "reward_total_composite_std": 0.1509392112493515} {"timestamp_utc": "2026-04-12T23:53:17Z", "mode": "train", "global_step": 696, "epoch": 0.06991461577096936, "loss": 0.0785, "grad_norm": 10.745696067810059, "learning_rate": 7.893939393939395e-06, "num_tokens": 1286802.0, "completions/mean_length": 112.75, "completions/min_length": 75.0, "completions/max_length": 185.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.75, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 185.0, "rewards/meter/mean": 0.9412195682525635, "rewards/meter/std": 0.1384810358285904, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9298588037490845, "rewards/repeat_soft/std": 0.10499134659767151, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.7561596632003784, "rewards/total_composite/std": 0.059901680797338486, "reward": 0.7561596632003784, "reward_std": 0.059901703149080276, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1558866947889328, "sampling/sampling_logp_difference/max": 1.4355297088623047, "sampling/importance_sampling_ratio/min": 0.23798927664756775, "sampling/importance_sampling_ratio/mean": 1.0264499187469482, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3603328987956047, "clip_ratio/low_mean": 0.027951981872320175, "clip_ratio/low_min": 0.027951981872320175, "clip_ratio/high_mean": 0.11671601887792349, "clip_ratio/high_max": 0.11671601887792349, "clip_ratio/region_mean": 0.14466800075024366, "reward_total_mean": 0.7561596632003784, "reward_meter_mean": 0.9412195682525635, "reward_meter_std": 0.1384810358285904, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9298588037490845, "reward_repeat_soft_std": 0.10499134659767151, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.7561596632003784, "reward_total_composite_std": 0.059901680797338486} {"timestamp_utc": "2026-04-12T23:53:24Z", "mode": "train", "global_step": 697, "epoch": 0.07001506780512305, "loss": 0.0405, "grad_norm": 21.028831481933594, "learning_rate": 7.89090909090909e-06, "num_tokens": 1288495.0, "completions/mean_length": 47.625, "completions/min_length": 38.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.625, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.8485350012779236, "rewards/meter/std": 0.24346700310707092, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9944108724594116, "rewards/repeat_soft/std": 0.0061992136761546135, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7595318555831909, "rewards/total_composite/std": 0.10702308267354965, "reward": 0.7595318555831909, "reward_std": 0.10702308267354965, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18772584199905396, "sampling/sampling_logp_difference/max": 1.4954845905303955, "sampling/importance_sampling_ratio/min": 0.2241399735212326, "sampling/importance_sampling_ratio/mean": 1.0138640403747559, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6234357953071594, "clip_ratio/low_mean": 0.028138297609984875, "clip_ratio/low_min": 0.028138297609984875, "clip_ratio/high_mean": 0.14525794237852097, "clip_ratio/high_max": 0.14525794237852097, "clip_ratio/region_mean": 0.17339623998850584, "reward_total_mean": 0.7595318555831909, "reward_meter_mean": 0.8485350012779236, "reward_meter_std": 0.24346700310707092, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9944108724594116, "reward_repeat_soft_std": 0.0061992136761546135, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7595318555831909, "reward_total_composite_std": 0.10702308267354965} {"timestamp_utc": "2026-04-12T23:53:32Z", "mode": "train", "global_step": 698, "epoch": 0.07011551983927675, "loss": 0.0754, "grad_norm": 14.939131736755371, "learning_rate": 7.88787878787879e-06, "num_tokens": 1290539.0, "completions/mean_length": 57.5, "completions/min_length": 45.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.5, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.8583699464797974, "rewards/meter/std": 0.3234589993953705, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.17817415297031403, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9811617136001587, "rewards/repeat_soft/std": 0.018229268491268158, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7290076017379761, "rewards/total_composite/std": 0.1554258018732071, "reward": 0.7290076017379761, "reward_std": 0.1554257869720459, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2065279632806778, "sampling/sampling_logp_difference/max": 2.232452392578125, "sampling/importance_sampling_ratio/min": 0.10726504772901535, "sampling/importance_sampling_ratio/mean": 1.0244697332382202, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4155068546533585, "clip_ratio/low_mean": 0.028985507786273956, "clip_ratio/low_min": 0.028985507786273956, "clip_ratio/high_mean": 0.13755581434816122, "clip_ratio/high_max": 0.13755581434816122, "clip_ratio/region_mean": 0.16654132213443518, "reward_total_mean": 0.7290076017379761, "reward_meter_mean": 0.8583699464797974, "reward_meter_std": 0.3234589993953705, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.17817415297031403, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9811617136001587, "reward_repeat_soft_std": 0.018229268491268158, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7290076017379761, "reward_total_composite_std": 0.1554258018732071} {"timestamp_utc": "2026-04-12T23:53:40Z", "mode": "train", "global_step": 699, "epoch": 0.07021597187343044, "loss": -0.0139, "grad_norm": 7.725651741027832, "learning_rate": 7.884848484848485e-06, "num_tokens": 1293301.0, "completions/mean_length": 110.25, "completions/min_length": 102.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.25, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.8537176251411438, "rewards/meter/std": 0.30518391728401184, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9982411861419678, "rewards/repeat_soft/std": 0.002757962327450514, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7299970388412476, "rewards/total_composite/std": 0.137348011136055, "reward": 0.7299970388412476, "reward_std": 0.137348011136055, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15909700095653534, "sampling/sampling_logp_difference/max": 1.6085309982299805, "sampling/importance_sampling_ratio/min": 0.20018146932125092, "sampling/importance_sampling_ratio/mean": 1.0201725959777832, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4324118420481682, "clip_ratio/low_mean": 0.018382353708148003, "clip_ratio/low_min": 0.018382353708148003, "clip_ratio/high_mean": 0.12242443673312664, "clip_ratio/high_max": 0.12242443673312664, "clip_ratio/region_mean": 0.14080679044127464, "reward_total_mean": 0.7299970388412476, "reward_meter_mean": 0.8537176251411438, "reward_meter_std": 0.30518391728401184, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9982411861419678, "reward_repeat_soft_std": 0.002757962327450514, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7299970388412476, "reward_total_composite_std": 0.137348011136055} {"timestamp_utc": "2026-04-12T23:53:47Z", "mode": "train", "global_step": 700, "epoch": 0.07031642390758412, "loss": 0.0093, "grad_norm": 10.670256614685059, "learning_rate": 7.881818181818182e-06, "num_tokens": 1295221.0, "completions/mean_length": 69.0, "completions/min_length": 54.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.9896233081817627, "rewards/meter/std": 0.00640181265771389, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8893686532974243, "rewards/repeat_soft/std": 0.12902189791202545, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.7776423692703247, "rewards/total_composite/std": 0.04690463840961456, "reward": 0.7776423692703247, "reward_std": 0.04690462723374367, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15415680408477783, "sampling/sampling_logp_difference/max": 1.8925161361694336, "sampling/importance_sampling_ratio/min": 0.1506921648979187, "sampling/importance_sampling_ratio/mean": 1.0370594263076782, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1441161185503006, "clip_ratio/low_mean": 0.05282248929142952, "clip_ratio/low_min": 0.05282248929142952, "clip_ratio/high_mean": 0.08624167926609516, "clip_ratio/high_max": 0.08624167926609516, "clip_ratio/region_mean": 0.13906416855752468, "reward_total_mean": 0.7776423692703247, "reward_meter_mean": 0.9896233081817627, "reward_meter_std": 0.00640181265771389, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8893686532974243, "reward_repeat_soft_std": 0.12902189791202545, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.7776423692703247, "reward_total_composite_std": 0.04690463840961456} {"timestamp_utc": "2026-04-12T23:54:49Z", "mode": "eval", "global_step": 700, "epoch": 0.07031642390758412, "eval_loss": NaN, "eval_runtime": 61.6524, "eval_samples_per_second": 1.298, "eval_steps_per_second": 0.162, "eval_num_tokens": 1295221.0, "eval_completions/mean_length": 68.65, "eval_completions/min_length": 28.6, "eval_completions/max_length": 150.6, "eval_completions/clipped_ratio": 0.0125, "eval_completions/mean_terminated_length": 63.04285736083985, "eval_completions/min_terminated_length": 28.6, "eval_completions/max_terminated_length": 113.8, "eval_rewards/meter/mean": 0.7379507124423981, "eval_rewards/meter/std": 0.2976392790675163, "eval_rewards/count_adherence/mean": 0.9264583170413971, "eval_rewards/count_adherence/std": 0.10502266883850098, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.046291005611419675, "eval_rewards/repeat_soft/mean": 0.9728111267089844, "eval_rewards/repeat_soft/std": 0.03835855443030596, "eval_rewards/judge_quality/mean": 0.45237499177455903, "eval_rewards/judge_quality/std": 0.16209346055984497, "eval_rewards/total_composite/mean": 0.693105673789978, "eval_rewards/total_composite/std": 0.1647371307015419, "eval_reward": 0.693105673789978, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.11498960256576538, "eval_sampling/sampling_logp_difference/max": 1.0311152458190918, "eval_sampling/importance_sampling_ratio/min": 0.3589024394750595, "eval_sampling/importance_sampling_ratio/mean": 1.0318241477012635, "eval_sampling/importance_sampling_ratio/max": 1.5449663519859314, "eval_entropy": 1.4189607143402099, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.693105673789978, "eval_reward_meter_mean": 0.7379507124423981, "eval_reward_meter_std": 0.2976392790675163, "eval_reward_count_adherence_mean": 0.9264583170413971, "eval_reward_count_adherence_std": 0.10502266883850098, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.046291005611419675, "eval_reward_repeat_soft_mean": 0.9728111267089844, "eval_reward_repeat_soft_std": 0.03835855443030596, "eval_reward_judge_quality_mean": 0.45237499177455903, "eval_reward_judge_quality_std": 0.16209346055984497, "eval_reward_total_composite_mean": 0.693105673789978, "eval_reward_total_composite_std": 0.1647371307015419} {"timestamp_utc": "2026-04-12T23:55:06Z", "mode": "train", "global_step": 701, "epoch": 0.07041687594173782, "loss": 0.0653, "grad_norm": 13.96159553527832, "learning_rate": 7.87878787878788e-06, "num_tokens": 1296849.0, "completions/mean_length": 37.5, "completions/min_length": 32.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.5, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.7107541561126709, "rewards/meter/std": 0.4216652512550354, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9980316162109375, "rewards/repeat_soft/std": 0.0032580101396888494, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.7342675924301147, "rewards/total_composite/std": 0.22649143636226654, "reward": 0.7342675924301147, "reward_std": 0.22649145126342773, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22207026183605194, "sampling/sampling_logp_difference/max": 2.2583718299865723, "sampling/importance_sampling_ratio/min": 0.1045205220580101, "sampling/importance_sampling_ratio/mean": 1.039129376411438, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7136566936969757, "clip_ratio/low_mean": 0.06887435913085938, "clip_ratio/low_min": 0.06887435913085938, "clip_ratio/high_mean": 0.1000829404219985, "clip_ratio/high_max": 0.1000829404219985, "clip_ratio/region_mean": 0.16895729955285788, "reward_total_mean": 0.7342675924301147, "reward_meter_mean": 0.7107541561126709, "reward_meter_std": 0.4216652512550354, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9980316162109375, "reward_repeat_soft_std": 0.0032580101396888494, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.7342675924301147, "reward_total_composite_std": 0.22649143636226654} {"timestamp_utc": "2026-04-12T23:55:15Z", "mode": "train", "global_step": 702, "epoch": 0.07051732797589151, "loss": 0.0805, "grad_norm": 11.713578224182129, "learning_rate": 7.875757575757577e-06, "num_tokens": 1298967.0, "completions/mean_length": 74.75, "completions/min_length": 67.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.75, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.9485121369361877, "rewards/meter/std": 0.03647454082965851, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9894789457321167, "rewards/repeat_soft/std": 0.0070917243137955666, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.7684033513069153, "rewards/total_composite/std": 0.02463621459901333, "reward": 0.7684033513069153, "reward_std": 0.024636197835206985, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20930412411689758, "sampling/sampling_logp_difference/max": 2.6858699321746826, "sampling/importance_sampling_ratio/min": 0.06816187500953674, "sampling/importance_sampling_ratio/mean": 1.0419529676437378, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.996992588043213, "clip_ratio/low_mean": 0.05542550981044769, "clip_ratio/low_min": 0.05542550981044769, "clip_ratio/high_mean": 0.10480602737516165, "clip_ratio/high_max": 0.10480602737516165, "clip_ratio/region_mean": 0.16023153718560934, "reward_total_mean": 0.7684033513069153, "reward_meter_mean": 0.9485121369361877, "reward_meter_std": 0.03647454082965851, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9894789457321167, "reward_repeat_soft_std": 0.0070917243137955666, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.7684033513069153, "reward_total_composite_std": 0.02463621459901333} {"timestamp_utc": "2026-04-12T23:55:25Z", "mode": "train", "global_step": 703, "epoch": 0.0706177800100452, "loss": 0.155, "grad_norm": 10.158346176147461, "learning_rate": 7.872727272727273e-06, "num_tokens": 1301175.0, "completions/mean_length": 84.0, "completions/min_length": 71.0, "completions/max_length": 147.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.0, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 147.0, "rewards/meter/mean": 0.8125337362289429, "rewards/meter/std": 0.28295084834098816, "rewards/count_adherence/mean": 0.7749999761581421, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9794973134994507, "rewards/repeat_soft/std": 0.04792679473757744, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.27523693442344666, "rewards/total_composite/mean": 0.7444648742675781, "rewards/total_composite/std": 0.13564273715019226, "reward": 0.7444648742675781, "reward_std": 0.13564275205135345, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13930577039718628, "sampling/sampling_logp_difference/max": 1.5838737487792969, "sampling/importance_sampling_ratio/min": 0.20517875254154205, "sampling/importance_sampling_ratio/mean": 1.0276713371276855, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2086697220802307, "clip_ratio/low_mean": 0.051270958967506886, "clip_ratio/low_min": 0.051270958967506886, "clip_ratio/high_mean": 0.09669754840433598, "clip_ratio/high_max": 0.09669754840433598, "clip_ratio/region_mean": 0.14796850737184286, "reward_total_mean": 0.7444648742675781, "reward_meter_mean": 0.8125337362289429, "reward_meter_std": 0.28295084834098816, "reward_count_adherence_mean": 0.7749999761581421, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9794973134994507, "reward_repeat_soft_std": 0.04792679473757744, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.27523693442344666, "reward_total_composite_mean": 0.7444648742675781, "reward_total_composite_std": 0.13564273715019226} {"timestamp_utc": "2026-04-12T23:55:33Z", "mode": "train", "global_step": 704, "epoch": 0.0707182320441989, "loss": 0.0775, "grad_norm": 19.1743221282959, "learning_rate": 7.86969696969697e-06, "num_tokens": 1303113.0, "completions/mean_length": 53.25, "completions/min_length": 45.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.25, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.4905164837837219, "rewards/meter/std": 0.3627164661884308, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9739729166030884, "rewards/repeat_soft/std": 0.04520575702190399, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.5941296815872192, "rewards/total_composite/std": 0.16455170512199402, "reward": 0.5941296815872192, "reward_std": 0.1645517200231552, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2074299454689026, "sampling/sampling_logp_difference/max": 2.283071279525757, "sampling/importance_sampling_ratio/min": 0.10197053849697113, "sampling/importance_sampling_ratio/mean": 0.9908467531204224, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1138741746544838, "clip_ratio/low_mean": 0.09932564944028854, "clip_ratio/low_min": 0.09932564944028854, "clip_ratio/high_mean": 0.07832198310643435, "clip_ratio/high_max": 0.07832198310643435, "clip_ratio/region_mean": 0.1776476325467229, "reward_total_mean": 0.5941296815872192, "reward_meter_mean": 0.4905164837837219, "reward_meter_std": 0.3627164661884308, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9739729166030884, "reward_repeat_soft_std": 0.04520575702190399, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.5941296815872192, "reward_total_composite_std": 0.16455170512199402} {"timestamp_utc": "2026-04-12T23:55:40Z", "mode": "train", "global_step": 705, "epoch": 0.07081868407835258, "loss": 0.05, "grad_norm": 15.10171127319336, "learning_rate": 7.866666666666667e-06, "num_tokens": 1305075.0, "completions/mean_length": 68.25, "completions/min_length": 50.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.25, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.5441829562187195, "rewards/meter/std": 0.32029563188552856, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9966331720352173, "rewards/repeat_soft/std": 0.0032992165070027113, "rewards/judge_quality/mean": 0.5699999928474426, "rewards/judge_quality/std": 0.16035676002502441, "rewards/total_composite/mean": 0.660858154296875, "rewards/total_composite/std": 0.17296136915683746, "reward": 0.660858154296875, "reward_std": 0.17296136915683746, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15422002971172333, "sampling/sampling_logp_difference/max": 2.6743202209472656, "sampling/importance_sampling_ratio/min": 0.06895368546247482, "sampling/importance_sampling_ratio/mean": 1.0175144672393799, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.017797365784645, "clip_ratio/low_mean": 0.08627345133572817, "clip_ratio/low_min": 0.08627345133572817, "clip_ratio/high_mean": 0.07728174794465303, "clip_ratio/high_max": 0.07728174794465303, "clip_ratio/region_mean": 0.1635551992803812, "reward_total_mean": 0.660858154296875, "reward_meter_mean": 0.5441829562187195, "reward_meter_std": 0.32029563188552856, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9966331720352173, "reward_repeat_soft_std": 0.0032992165070027113, "reward_judge_quality_mean": 0.5699999928474426, "reward_judge_quality_std": 0.16035676002502441, "reward_total_composite_mean": 0.660858154296875, "reward_total_composite_std": 0.17296136915683746} {"timestamp_utc": "2026-04-12T23:55:50Z", "mode": "train", "global_step": 706, "epoch": 0.07091913611250628, "loss": 0.0676, "grad_norm": 11.824435234069824, "learning_rate": 7.863636363636364e-06, "num_tokens": 1307033.0, "completions/mean_length": 75.75, "completions/min_length": 63.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.75, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.5866546034812927, "rewards/meter/std": 0.31946635246276855, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9690037369728088, "rewards/repeat_soft/std": 0.023960642516613007, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.624269962310791, "rewards/total_composite/std": 0.14560241997241974, "reward": 0.624269962310791, "reward_std": 0.14560240507125854, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20899225771427155, "sampling/sampling_logp_difference/max": 1.3694686889648438, "sampling/importance_sampling_ratio/min": 0.2542420029640198, "sampling/importance_sampling_ratio/mean": 1.0535465478897095, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2604872584342957, "clip_ratio/low_mean": 0.09283924661576748, "clip_ratio/low_min": 0.09283924661576748, "clip_ratio/high_mean": 0.08953955210745335, "clip_ratio/high_max": 0.08953955210745335, "clip_ratio/region_mean": 0.18237879872322083, "reward_total_mean": 0.624269962310791, "reward_meter_mean": 0.5866546034812927, "reward_meter_std": 0.31946635246276855, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9690037369728088, "reward_repeat_soft_std": 0.023960642516613007, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.624269962310791, "reward_total_composite_std": 0.14560241997241974} {"timestamp_utc": "2026-04-12T23:55:59Z", "mode": "train", "global_step": 707, "epoch": 0.07101958814665997, "loss": 0.1433, "grad_norm": 8.44680118560791, "learning_rate": 7.860606060606062e-06, "num_tokens": 1309811.0, "completions/mean_length": 146.25, "completions/min_length": 114.0, "completions/max_length": 203.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 146.25, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 203.0, "rewards/meter/mean": 0.9658769369125366, "rewards/meter/std": 0.04289662837982178, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9561344385147095, "rewards/repeat_soft/std": 0.025806451216340065, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7998830676078796, "rewards/total_composite/std": 0.03675498440861702, "reward": 0.7998830676078796, "reward_std": 0.03675498068332672, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1894703358411789, "sampling/sampling_logp_difference/max": 1.7396769523620605, "sampling/importance_sampling_ratio/min": 0.17557711899280548, "sampling/importance_sampling_ratio/mean": 1.03104829788208, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5541511625051498, "clip_ratio/low_mean": 0.04181594401597977, "clip_ratio/low_min": 0.04181594401597977, "clip_ratio/high_mean": 0.12544529512524605, "clip_ratio/high_max": 0.12544529512524605, "clip_ratio/region_mean": 0.16726123914122581, "reward_total_mean": 0.7998830676078796, "reward_meter_mean": 0.9658769369125366, "reward_meter_std": 0.04289662837982178, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9561344385147095, "reward_repeat_soft_std": 0.025806451216340065, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7998830676078796, "reward_total_composite_std": 0.03675498440861702} {"timestamp_utc": "2026-04-12T23:56:07Z", "mode": "train", "global_step": 708, "epoch": 0.07112004018081367, "loss": 0.0877, "grad_norm": 15.55359935760498, "learning_rate": 7.857575757575759e-06, "num_tokens": 1311542.0, "completions/mean_length": 48.375, "completions/min_length": 40.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.375, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.4671587347984314, "rewards/meter/std": 0.4299677908420563, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9683951139450073, "rewards/repeat_soft/std": 0.01767519675195217, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.5568109750747681, "rewards/total_composite/std": 0.18362294137477875, "reward": 0.5568109750747681, "reward_std": 0.18362291157245636, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18954430520534515, "sampling/sampling_logp_difference/max": 1.4494256973266602, "sampling/importance_sampling_ratio/min": 0.23470506072044373, "sampling/importance_sampling_ratio/mean": 1.01266610622406, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6708464920520782, "clip_ratio/low_mean": 0.07391549553722143, "clip_ratio/low_min": 0.07391549553722143, "clip_ratio/high_mean": 0.09250000305473804, "clip_ratio/high_max": 0.09250000305473804, "clip_ratio/region_mean": 0.16641549859195948, "reward_total_mean": 0.5568109750747681, "reward_meter_mean": 0.4671587347984314, "reward_meter_std": 0.4299677908420563, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9683951139450073, "reward_repeat_soft_std": 0.01767519675195217, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.5568109750747681, "reward_total_composite_std": 0.18362294137477875} {"timestamp_utc": "2026-04-12T23:56:13Z", "mode": "train", "global_step": 709, "epoch": 0.07122049221496736, "loss": -0.003, "grad_norm": 15.373880386352539, "learning_rate": 7.854545454545454e-06, "num_tokens": 1313038.0, "completions/mean_length": 37.0, "completions/min_length": 32.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.2439814954996109, "rewards/meter/std": 0.2028210312128067, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.984288215637207, "rewards/repeat_soft/std": 0.023304061964154243, "rewards/judge_quality/mean": 0.5649999976158142, "rewards/judge_quality/std": 0.20057062804698944, "rewards/total_composite/mean": 0.5277204513549805, "rewards/total_composite/std": 0.11870817095041275, "reward": 0.5277204513549805, "reward_std": 0.11870817095041275, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16117635369300842, "sampling/sampling_logp_difference/max": 1.4032106399536133, "sampling/importance_sampling_ratio/min": 0.2458065003156662, "sampling/importance_sampling_ratio/mean": 1.0096536874771118, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1998685523867607, "clip_ratio/low_mean": 0.06733631156384945, "clip_ratio/low_min": 0.06733631156384945, "clip_ratio/high_mean": 0.07725048251450062, "clip_ratio/high_max": 0.07725048251450062, "clip_ratio/region_mean": 0.14458679407835007, "reward_total_mean": 0.5277204513549805, "reward_meter_mean": 0.2439814954996109, "reward_meter_std": 0.2028210312128067, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.984288215637207, "reward_repeat_soft_std": 0.023304061964154243, "reward_judge_quality_mean": 0.5649999976158142, "reward_judge_quality_std": 0.20057062804698944, "reward_total_composite_mean": 0.5277204513549805, "reward_total_composite_std": 0.11870817095041275} {"timestamp_utc": "2026-04-12T23:56:21Z", "mode": "train", "global_step": 710, "epoch": 0.07132094424912104, "loss": 0.0009, "grad_norm": 14.736470222473145, "learning_rate": 7.851515151515152e-06, "num_tokens": 1314613.0, "completions/mean_length": 33.875, "completions/min_length": 27.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.875, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.7543967962265015, "rewards/meter/std": 0.3606851100921631, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9875748157501221, "rewards/repeat_soft/std": 0.02182442881166935, "rewards/judge_quality/mean": 0.5862500071525574, "rewards/judge_quality/std": 0.22984081506729126, "rewards/total_composite/mean": 0.6603624820709229, "rewards/total_composite/std": 0.29670360684394836, "reward": 0.6603624820709229, "reward_std": 0.29670360684394836, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1911056637763977, "sampling/sampling_logp_difference/max": 1.2623953819274902, "sampling/importance_sampling_ratio/min": 0.28297537565231323, "sampling/importance_sampling_ratio/mean": 1.0442181825637817, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7052520513534546, "clip_ratio/low_mean": 0.07066634111106396, "clip_ratio/low_min": 0.07066634111106396, "clip_ratio/high_mean": 0.13620463386178017, "clip_ratio/high_max": 0.13620463386178017, "clip_ratio/region_mean": 0.20687097497284412, "reward_total_mean": 0.6603624820709229, "reward_meter_mean": 0.7543967962265015, "reward_meter_std": 0.3606851100921631, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9875748157501221, "reward_repeat_soft_std": 0.02182442881166935, "reward_judge_quality_mean": 0.5862500071525574, "reward_judge_quality_std": 0.22984081506729126, "reward_total_composite_mean": 0.6603624820709229, "reward_total_composite_std": 0.29670360684394836} {"timestamp_utc": "2026-04-12T23:56:28Z", "mode": "train", "global_step": 711, "epoch": 0.07142139628327474, "loss": -0.028, "grad_norm": 13.684534072875977, "learning_rate": 7.848484848484849e-06, "num_tokens": 1316246.0, "completions/mean_length": 43.125, "completions/min_length": 36.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.125, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.7687084674835205, "rewards/meter/std": 0.3685309588909149, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9797462224960327, "rewards/repeat_soft/std": 0.02363337203860283, "rewards/judge_quality/mean": 0.59375, "rewards/judge_quality/std": 0.2775370180606842, "rewards/total_composite/mean": 0.7278727293014526, "rewards/total_composite/std": 0.3113797605037689, "reward": 0.7278727293014526, "reward_std": 0.3113797903060913, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18120528757572174, "sampling/sampling_logp_difference/max": 1.708449363708496, "sampling/importance_sampling_ratio/min": 0.18114645779132843, "sampling/importance_sampling_ratio/mean": 1.0169594287872314, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4436416774988174, "clip_ratio/low_mean": 0.0325573543086648, "clip_ratio/low_min": 0.0325573543086648, "clip_ratio/high_mean": 0.1347047844901681, "clip_ratio/high_max": 0.1347047844901681, "clip_ratio/region_mean": 0.1672621387988329, "reward_total_mean": 0.7278727293014526, "reward_meter_mean": 0.7687084674835205, "reward_meter_std": 0.3685309588909149, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9797462224960327, "reward_repeat_soft_std": 0.02363337203860283, "reward_judge_quality_mean": 0.59375, "reward_judge_quality_std": 0.2775370180606842, "reward_total_composite_mean": 0.7278727293014526, "reward_total_composite_std": 0.3113797605037689} {"timestamp_utc": "2026-04-12T23:56:35Z", "mode": "train", "global_step": 712, "epoch": 0.07152184831742843, "loss": 0.0351, "grad_norm": 13.2311372756958, "learning_rate": 7.845454545454546e-06, "num_tokens": 1317765.0, "completions/mean_length": 41.875, "completions/min_length": 37.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.875, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.42368894815444946, "rewards/meter/std": 0.3684481382369995, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.945104718208313, "rewards/repeat_soft/std": 0.08772008121013641, "rewards/judge_quality/mean": 0.44875001907348633, "rewards/judge_quality/std": 0.21256513893604279, "rewards/total_composite/mean": 0.5697954893112183, "rewards/total_composite/std": 0.15912161767482758, "reward": 0.5697954893112183, "reward_std": 0.15912163257598877, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14881440997123718, "sampling/sampling_logp_difference/max": 1.0267260074615479, "sampling/importance_sampling_ratio/min": 0.3864566683769226, "sampling/importance_sampling_ratio/mean": 1.0496243238449097, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2565553858876228, "clip_ratio/low_mean": 0.11254037730395794, "clip_ratio/low_min": 0.11254037730395794, "clip_ratio/high_mean": 0.039843445643782616, "clip_ratio/high_max": 0.039843445643782616, "clip_ratio/region_mean": 0.15238382294774055, "reward_total_mean": 0.5697954893112183, "reward_meter_mean": 0.42368894815444946, "reward_meter_std": 0.3684481382369995, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.945104718208313, "reward_repeat_soft_std": 0.08772008121013641, "reward_judge_quality_mean": 0.44875001907348633, "reward_judge_quality_std": 0.21256513893604279, "reward_total_composite_mean": 0.5697954893112183, "reward_total_composite_std": 0.15912161767482758} {"timestamp_utc": "2026-04-12T23:56:41Z", "mode": "train", "global_step": 713, "epoch": 0.07162230035158212, "loss": 0.0966, "grad_norm": 17.272686004638672, "learning_rate": 7.842424242424243e-06, "num_tokens": 1319325.0, "completions/mean_length": 36.0, "completions/min_length": 32.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.8957808613777161, "rewards/meter/std": 0.24228346347808838, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.959288477897644, "rewards/repeat_soft/std": 0.025506792590022087, "rewards/judge_quality/mean": 0.8700000047683716, "rewards/judge_quality/std": 0.09258200973272324, "rewards/total_composite/mean": 0.9100302457809448, "rewards/total_composite/std": 0.10525497794151306, "reward": 0.9100302457809448, "reward_std": 0.10525497049093246, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1259698122739792, "sampling/sampling_logp_difference/max": 1.7169773578643799, "sampling/importance_sampling_ratio/min": 0.17960822582244873, "sampling/importance_sampling_ratio/mean": 1.0190012454986572, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7730044946074486, "clip_ratio/low_mean": 0.03464330779388547, "clip_ratio/low_min": 0.03464330779388547, "clip_ratio/high_mean": 0.05991308041848242, "clip_ratio/high_max": 0.05991308041848242, "clip_ratio/region_mean": 0.09455638821236789, "reward_total_mean": 0.9100302457809448, "reward_meter_mean": 0.8957808613777161, "reward_meter_std": 0.24228346347808838, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.959288477897644, "reward_repeat_soft_std": 0.025506792590022087, "reward_judge_quality_mean": 0.8700000047683716, "reward_judge_quality_std": 0.09258200973272324, "reward_total_composite_mean": 0.9100302457809448, "reward_total_composite_std": 0.10525497794151306} {"timestamp_utc": "2026-04-12T23:56:47Z", "mode": "train", "global_step": 714, "epoch": 0.0717227523857358, "loss": 0.014, "grad_norm": 19.554704666137695, "learning_rate": 7.83939393939394e-06, "num_tokens": 1320766.0, "completions/mean_length": 20.125, "completions/min_length": 18.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.125, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.7527880072593689, "rewards/meter/std": 0.4004596471786499, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.3962499797344208, "rewards/judge_quality/std": 0.09085899591445923, "rewards/total_composite/mean": 0.6114614605903625, "rewards/total_composite/std": 0.3010958135128021, "reward": 0.6114614605903625, "reward_std": 0.30109578371047974, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1792704164981842, "sampling/sampling_logp_difference/max": 2.485342502593994, "sampling/importance_sampling_ratio/min": 0.08329702168703079, "sampling/importance_sampling_ratio/mean": 1.0327544212341309, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3274625837802887, "clip_ratio/low_mean": 0.056746033020317554, "clip_ratio/low_min": 0.056746033020317554, "clip_ratio/high_mean": 0.12804206274449825, "clip_ratio/high_max": 0.12804206274449825, "clip_ratio/region_mean": 0.1847880957648158, "reward_total_mean": 0.6114614605903625, "reward_meter_mean": 0.7527880072593689, "reward_meter_std": 0.4004596471786499, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.3962499797344208, "reward_judge_quality_std": 0.09085899591445923, "reward_total_composite_mean": 0.6114614605903625, "reward_total_composite_std": 0.3010958135128021} {"timestamp_utc": "2026-04-12T23:56:58Z", "mode": "train", "global_step": 715, "epoch": 0.0718232044198895, "loss": -0.1249, "grad_norm": 3.80837345123291, "learning_rate": 7.836363636363638e-06, "num_tokens": 1322540.0, "completions/mean_length": 121.75, "completions/min_length": 55.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.5173114538192749, "rewards/meter/std": 0.38818904757499695, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9907404184341431, "rewards/repeat_soft/std": 0.008140078745782375, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.13274572789669037, "rewards/total_composite/mean": 0.5263596177101135, "rewards/total_composite/std": 0.26285871863365173, "reward": 0.5263596177101135, "reward_std": 0.26285871863365173, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18944275379180908, "sampling/sampling_logp_difference/max": 1.6122112274169922, "sampling/importance_sampling_ratio/min": 0.19944611191749573, "sampling/importance_sampling_ratio/mean": 1.027300238609314, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4834277033805847, "clip_ratio/low_mean": 0.05151211004704237, "clip_ratio/low_min": 0.05151211004704237, "clip_ratio/high_mean": 0.08458654768764973, "clip_ratio/high_max": 0.08458654768764973, "clip_ratio/region_mean": 0.1360986577346921, "reward_total_mean": 0.5263596177101135, "reward_meter_mean": 0.5173114538192749, "reward_meter_std": 0.38818904757499695, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9907404184341431, "reward_repeat_soft_std": 0.008140078745782375, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.13274572789669037, "reward_total_composite_mean": 0.5263596177101135, "reward_total_composite_std": 0.26285871863365173} {"timestamp_utc": "2026-04-12T23:57:12Z", "mode": "train", "global_step": 716, "epoch": 0.0719236564540432, "loss": -0.0288, "grad_norm": 6.755016326904297, "learning_rate": 7.833333333333333e-06, "num_tokens": 1324363.0, "completions/mean_length": 115.875, "completions/min_length": 45.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 59.28571701049805, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.5403679609298706, "rewards/meter/std": 0.3120501935482025, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9873404502868652, "rewards/repeat_soft/std": 0.012635891325771809, "rewards/judge_quality/mean": 0.3375000059604645, "rewards/judge_quality/std": 0.16722525656223297, "rewards/total_composite/mean": 0.4696003794670105, "rewards/total_composite/std": 0.30696654319763184, "reward": 0.4696003794670105, "reward_std": 0.30696651339530945, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17444925010204315, "sampling/sampling_logp_difference/max": 2.0181350708007812, "sampling/importance_sampling_ratio/min": 0.1329030990600586, "sampling/importance_sampling_ratio/mean": 1.0031895637512207, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5264361798763275, "clip_ratio/low_mean": 0.024383079260587692, "clip_ratio/low_min": 0.024383079260587692, "clip_ratio/high_mean": 0.12464602291584015, "clip_ratio/high_max": 0.12464602291584015, "clip_ratio/region_mean": 0.14902910217642784, "reward_total_mean": 0.4696003794670105, "reward_meter_mean": 0.5403679609298706, "reward_meter_std": 0.3120501935482025, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9873404502868652, "reward_repeat_soft_std": 0.012635891325771809, "reward_judge_quality_mean": 0.3375000059604645, "reward_judge_quality_std": 0.16722525656223297, "reward_total_composite_mean": 0.4696003794670105, "reward_total_composite_std": 0.30696654319763184} {"timestamp_utc": "2026-04-12T23:57:26Z", "mode": "train", "global_step": 717, "epoch": 0.07202410848819689, "loss": -0.101, "grad_norm": 5.362733364105225, "learning_rate": 7.83030303030303e-06, "num_tokens": 1326188.0, "completions/mean_length": 119.125, "completions/min_length": 50.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 63.000003814697266, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.421663761138916, "rewards/meter/std": 0.312870055437088, "rewards/count_adherence/mean": 0.71875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.993672251701355, "rewards/repeat_soft/std": 0.007648526690900326, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.23445606231689453, "rewards/total_composite/mean": 0.4452073276042938, "rewards/total_composite/std": 0.29019033908843994, "reward": 0.4452073276042938, "reward_std": 0.29019030928611755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2180832177400589, "sampling/sampling_logp_difference/max": 1.5292062759399414, "sampling/importance_sampling_ratio/min": 0.21670761704444885, "sampling/importance_sampling_ratio/mean": 1.0318816900253296, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9920809864997864, "clip_ratio/low_mean": 0.01886792480945587, "clip_ratio/low_min": 0.01886792480945587, "clip_ratio/high_mean": 0.15960875898599625, "clip_ratio/high_max": 0.15960875898599625, "clip_ratio/region_mean": 0.17847668379545212, "reward_total_mean": 0.4452073276042938, "reward_meter_mean": 0.421663761138916, "reward_meter_std": 0.312870055437088, "reward_count_adherence_mean": 0.71875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.993672251701355, "reward_repeat_soft_std": 0.007648526690900326, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.23445606231689453, "reward_total_composite_mean": 0.4452073276042938, "reward_total_composite_std": 0.29019033908843994} {"timestamp_utc": "2026-04-12T23:57:33Z", "mode": "train", "global_step": 718, "epoch": 0.07212456052235058, "loss": 0.0639, "grad_norm": 12.949377059936523, "learning_rate": 7.827272727272728e-06, "num_tokens": 1327863.0, "completions/mean_length": 48.375, "completions/min_length": 44.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.375, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.843619704246521, "rewards/meter/std": 0.22348423302173615, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9845238327980042, "rewards/repeat_soft/std": 0.025153663009405136, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.1974073201417923, "rewards/total_composite/mean": 0.7180320024490356, "rewards/total_composite/std": 0.29900500178337097, "reward": 0.7180320024490356, "reward_std": 0.29900500178337097, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1598617285490036, "sampling/sampling_logp_difference/max": 1.1451091766357422, "sampling/importance_sampling_ratio/min": 0.31818917393684387, "sampling/importance_sampling_ratio/mean": 1.0185508728027344, "sampling/importance_sampling_ratio/max": 1.9965418577194214, "entropy": 1.6304346472024918, "clip_ratio/low_mean": 0.034589314833283424, "clip_ratio/low_min": 0.034589314833283424, "clip_ratio/high_mean": 0.14455093070864677, "clip_ratio/high_max": 0.14455093070864677, "clip_ratio/region_mean": 0.1791402455419302, "reward_total_mean": 0.7180320024490356, "reward_meter_mean": 0.843619704246521, "reward_meter_std": 0.22348423302173615, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9845238327980042, "reward_repeat_soft_std": 0.025153663009405136, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.1974073201417923, "reward_total_composite_mean": 0.7180320024490356, "reward_total_composite_std": 0.29900500178337097} {"timestamp_utc": "2026-04-12T23:57:39Z", "mode": "train", "global_step": 719, "epoch": 0.07222501255650426, "loss": 0.0855, "grad_norm": 21.968605041503906, "learning_rate": 7.824242424242425e-06, "num_tokens": 1329464.0, "completions/mean_length": 42.125, "completions/min_length": 33.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.125, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.8329532146453857, "rewards/meter/std": 0.33181607723236084, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9885250329971313, "rewards/repeat_soft/std": 0.018283916637301445, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.1940544992685318, "rewards/total_composite/mean": 0.7631814479827881, "rewards/total_composite/std": 0.1660802811384201, "reward": 0.7631814479827881, "reward_std": 0.1660802662372589, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2035435289144516, "sampling/sampling_logp_difference/max": 1.3744583129882812, "sampling/importance_sampling_ratio/min": 0.25297659635543823, "sampling/importance_sampling_ratio/mean": 1.025065302848816, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9136250615119934, "clip_ratio/low_mean": 0.059230769984424114, "clip_ratio/low_min": 0.059230769984424114, "clip_ratio/high_mean": 0.15873290598392487, "clip_ratio/high_max": 0.15873290598392487, "clip_ratio/region_mean": 0.21796367596834898, "reward_total_mean": 0.7631814479827881, "reward_meter_mean": 0.8329532146453857, "reward_meter_std": 0.33181607723236084, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9885250329971313, "reward_repeat_soft_std": 0.018283916637301445, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.1940544992685318, "reward_total_composite_mean": 0.7631814479827881, "reward_total_composite_std": 0.1660802811384201} {"timestamp_utc": "2026-04-12T23:57:45Z", "mode": "train", "global_step": 720, "epoch": 0.07232546459065796, "loss": -0.0665, "grad_norm": 16.519514083862305, "learning_rate": 7.821212121212122e-06, "num_tokens": 1330986.0, "completions/mean_length": 24.25, "completions/min_length": 18.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.25, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.845525860786438, "rewards/meter/std": 0.327902227640152, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9371522665023804, "rewards/repeat_soft/std": 0.05339570343494415, "rewards/judge_quality/mean": 0.25, "rewards/judge_quality/std": 0.09258200228214264, "rewards/total_composite/mean": 0.6992018222808838, "rewards/total_composite/std": 0.1508513242006302, "reward": 0.6992018222808838, "reward_std": 0.15085133910179138, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15902215242385864, "sampling/sampling_logp_difference/max": 1.2152223587036133, "sampling/importance_sampling_ratio/min": 0.29664406180381775, "sampling/importance_sampling_ratio/mean": 1.0543131828308105, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1337349936366081, "clip_ratio/low_mean": 0.02083333395421505, "clip_ratio/low_min": 0.02083333395421505, "clip_ratio/high_mean": 0.13445475045591593, "clip_ratio/high_max": 0.13445475045591593, "clip_ratio/region_mean": 0.15528808441013098, "reward_total_mean": 0.6992018222808838, "reward_meter_mean": 0.845525860786438, "reward_meter_std": 0.327902227640152, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9371522665023804, "reward_repeat_soft_std": 0.05339570343494415, "reward_judge_quality_mean": 0.25, "reward_judge_quality_std": 0.09258200228214264, "reward_total_composite_mean": 0.6992018222808838, "reward_total_composite_std": 0.1508513242006302} {"timestamp_utc": "2026-04-12T23:57:57Z", "mode": "train", "global_step": 721, "epoch": 0.07242591662481165, "loss": -0.0893, "grad_norm": 5.538954734802246, "learning_rate": 7.81818181818182e-06, "num_tokens": 1332919.0, "completions/mean_length": 129.625, "completions/min_length": 64.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 75.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.6093370914459229, "rewards/meter/std": 0.4042149782180786, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9807107448577881, "rewards/repeat_soft/std": 0.02035585604608059, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.22211645543575287, "rewards/total_composite/mean": 0.5466623306274414, "rewards/total_composite/std": 0.34343039989471436, "reward": 0.5466623306274414, "reward_std": 0.34343039989471436, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21634259819984436, "sampling/sampling_logp_difference/max": 1.7141437530517578, "sampling/importance_sampling_ratio/min": 0.1801178902387619, "sampling/importance_sampling_ratio/mean": 1.002867579460144, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7029853165149689, "clip_ratio/low_mean": 0.0259146336466074, "clip_ratio/low_min": 0.0259146336466074, "clip_ratio/high_mean": 0.18014147132635117, "clip_ratio/high_max": 0.18014147132635117, "clip_ratio/region_mean": 0.20605610497295856, "reward_total_mean": 0.5466623306274414, "reward_meter_mean": 0.6093370914459229, "reward_meter_std": 0.4042149782180786, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9807107448577881, "reward_repeat_soft_std": 0.02035585604608059, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.22211645543575287, "reward_total_composite_mean": 0.5466623306274414, "reward_total_composite_std": 0.34343039989471436} {"timestamp_utc": "2026-04-12T23:58:09Z", "mode": "train", "global_step": 722, "epoch": 0.07252636865896535, "loss": -0.1218, "grad_norm": 4.380364894866943, "learning_rate": 7.815151515151515e-06, "num_tokens": 1334841.0, "completions/mean_length": 127.25, "completions/min_length": 47.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 72.28572082519531, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9804401397705078, "rewards/meter/std": 0.026022514328360558, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.17251639068126678, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9405562877655029, "rewards/repeat_soft/std": 0.12277265638113022, "rewards/judge_quality/mean": 0.41874998807907104, "rewards/judge_quality/std": 0.18074746429920197, "rewards/total_composite/mean": 0.6102756261825562, "rewards/total_composite/std": 0.3798506259918213, "reward": 0.6102756261825562, "reward_std": 0.3798506557941437, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17565231025218964, "sampling/sampling_logp_difference/max": 1.817692756652832, "sampling/importance_sampling_ratio/min": 0.162400022149086, "sampling/importance_sampling_ratio/mean": 1.0231083631515503, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.385721780359745, "clip_ratio/low_mean": 0.024193547666072845, "clip_ratio/low_min": 0.024193547666072845, "clip_ratio/high_mean": 0.10930593498051167, "clip_ratio/high_max": 0.10930593498051167, "clip_ratio/region_mean": 0.1334994826465845, "reward_total_mean": 0.6102756261825562, "reward_meter_mean": 0.9804401397705078, "reward_meter_std": 0.026022514328360558, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.17251639068126678, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9405562877655029, "reward_repeat_soft_std": 0.12277265638113022, "reward_judge_quality_mean": 0.41874998807907104, "reward_judge_quality_std": 0.18074746429920197, "reward_total_composite_mean": 0.6102756261825562, "reward_total_composite_std": 0.3798506259918213} {"timestamp_utc": "2026-04-12T23:58:20Z", "mode": "train", "global_step": 723, "epoch": 0.07262682069311903, "loss": -0.0907, "grad_norm": 4.859267711639404, "learning_rate": 7.812121212121213e-06, "num_tokens": 1336340.0, "completions/mean_length": 99.375, "completions/min_length": 33.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 40.42857360839844, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.6611332893371582, "rewards/meter/std": 0.40163007378578186, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9881571531295776, "rewards/repeat_soft/std": 0.01572253182530403, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.6162692308425903, "rewards/total_composite/std": 0.2940056622028351, "reward": 0.6162692308425903, "reward_std": 0.2940056622028351, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21043092012405396, "sampling/sampling_logp_difference/max": 2.4600276947021484, "sampling/importance_sampling_ratio/min": 0.21800647675991058, "sampling/importance_sampling_ratio/mean": 1.033679723739624, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8999276757240295, "clip_ratio/low_mean": 0.04189857840538025, "clip_ratio/low_min": 0.04189857840538025, "clip_ratio/high_mean": 0.13294764701277018, "clip_ratio/high_max": 0.13294764701277018, "clip_ratio/region_mean": 0.17484622541815042, "reward_total_mean": 0.6162692308425903, "reward_meter_mean": 0.6611332893371582, "reward_meter_std": 0.40163007378578186, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9881571531295776, "reward_repeat_soft_std": 0.01572253182530403, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.6162692308425903, "reward_total_composite_std": 0.2940056622028351} {"timestamp_utc": "2026-04-12T23:58:27Z", "mode": "train", "global_step": 724, "epoch": 0.07272727272727272, "loss": -0.0774, "grad_norm": 15.033525466918945, "learning_rate": 7.80909090909091e-06, "num_tokens": 1337908.0, "completions/mean_length": 25.0, "completions/min_length": 18.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.0, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.8526021242141724, "rewards/meter/std": 0.34302523732185364, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9609359502792358, "rewards/repeat_soft/std": 0.0034086306113749743, "rewards/judge_quality/mean": 0.2587500214576721, "rewards/judge_quality/std": 0.15037453174591064, "rewards/total_composite/mean": 0.7073895335197449, "rewards/total_composite/std": 0.17348907887935638, "reward": 0.7073895335197449, "reward_std": 0.17348907887935638, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13274243474006653, "sampling/sampling_logp_difference/max": 1.4573144912719727, "sampling/importance_sampling_ratio/min": 0.23286078870296478, "sampling/importance_sampling_ratio/mean": 0.9996877908706665, "sampling/importance_sampling_ratio/max": 1.6313120126724243, "entropy": 0.9733536317944527, "clip_ratio/low_mean": 0.0347222238779068, "clip_ratio/low_min": 0.0347222238779068, "clip_ratio/high_mean": 0.12411165330559015, "clip_ratio/high_max": 0.12411165330559015, "clip_ratio/region_mean": 0.15883387718349695, "reward_total_mean": 0.7073895335197449, "reward_meter_mean": 0.8526021242141724, "reward_meter_std": 0.34302523732185364, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9609359502792358, "reward_repeat_soft_std": 0.0034086306113749743, "reward_judge_quality_mean": 0.2587500214576721, "reward_judge_quality_std": 0.15037453174591064, "reward_total_composite_mean": 0.7073895335197449, "reward_total_composite_std": 0.17348907887935638} {"timestamp_utc": "2026-04-12T23:58:34Z", "mode": "train", "global_step": 725, "epoch": 0.07282772476142642, "loss": 0.0031, "grad_norm": 14.51097583770752, "learning_rate": 7.806060606060607e-06, "num_tokens": 1339786.0, "completions/mean_length": 52.75, "completions/min_length": 43.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.75, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.4777289032936096, "rewards/meter/std": 0.31828656792640686, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9894975423812866, "rewards/repeat_soft/std": 0.010725805535912514, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.52337646484375, "rewards/total_composite/std": 0.26658275723457336, "reward": 0.52337646484375, "reward_std": 0.26658275723457336, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20296135544776917, "sampling/sampling_logp_difference/max": 2.122990608215332, "sampling/importance_sampling_ratio/min": 0.11967319995164871, "sampling/importance_sampling_ratio/mean": 1.016248106956482, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1572262346744537, "clip_ratio/low_mean": 0.09468726813793182, "clip_ratio/low_min": 0.09468726813793182, "clip_ratio/high_mean": 0.09461412765085697, "clip_ratio/high_max": 0.09461412765085697, "clip_ratio/region_mean": 0.1893013957887888, "reward_total_mean": 0.52337646484375, "reward_meter_mean": 0.4777289032936096, "reward_meter_std": 0.31828656792640686, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9894975423812866, "reward_repeat_soft_std": 0.010725805535912514, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.52337646484375, "reward_total_composite_std": 0.26658275723457336} {"timestamp_utc": "2026-04-12T23:58:47Z", "mode": "train", "global_step": 726, "epoch": 0.07292817679558011, "loss": -0.0755, "grad_norm": 5.218565464019775, "learning_rate": 7.803030303030303e-06, "num_tokens": 1341670.0, "completions/mean_length": 128.5, "completions/min_length": 62.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 73.71428680419922, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.9213427901268005, "rewards/meter/std": 0.057977933436632156, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9885221719741821, "rewards/repeat_soft/std": 0.010410452261567116, "rewards/judge_quality/mean": 0.44874998927116394, "rewards/judge_quality/std": 0.2105392962694168, "rewards/total_composite/mean": 0.7680814266204834, "rewards/total_composite/std": 0.0777624100446701, "reward": 0.7680814266204834, "reward_std": 0.07776240259408951, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1879929006099701, "sampling/sampling_logp_difference/max": 1.2643595933914185, "sampling/importance_sampling_ratio/min": 0.2824200987815857, "sampling/importance_sampling_ratio/mean": 1.0384306907653809, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6587970852851868, "clip_ratio/low_mean": 0.060769825242459774, "clip_ratio/low_min": 0.060769825242459774, "clip_ratio/high_mean": 0.0913188997656107, "clip_ratio/high_max": 0.0913188997656107, "clip_ratio/region_mean": 0.15208872500807047, "reward_total_mean": 0.7680814266204834, "reward_meter_mean": 0.9213427901268005, "reward_meter_std": 0.057977933436632156, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9885221719741821, "reward_repeat_soft_std": 0.010410452261567116, "reward_judge_quality_mean": 0.44874998927116394, "reward_judge_quality_std": 0.2105392962694168, "reward_total_composite_mean": 0.7680814266204834, "reward_total_composite_std": 0.0777624100446701} {"timestamp_utc": "2026-04-12T23:59:00Z", "mode": "train", "global_step": 727, "epoch": 0.07302862882973381, "loss": -0.1257, "grad_norm": 2.360023260116577, "learning_rate": 7.800000000000002e-06, "num_tokens": 1343425.0, "completions/mean_length": 103.375, "completions/min_length": 40.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 45.000003814697266, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.9107831716537476, "rewards/meter/std": 0.10744628310203552, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9722753763198853, "rewards/repeat_soft/std": 0.03736523538827896, "rewards/judge_quality/mean": 0.6899999976158142, "rewards/judge_quality/std": 0.28400203585624695, "rewards/total_composite/mean": 0.7880845665931702, "rewards/total_composite/std": 0.32319119572639465, "reward": 0.7880845665931702, "reward_std": 0.32319119572639465, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20278142392635345, "sampling/sampling_logp_difference/max": 1.3290700912475586, "sampling/importance_sampling_ratio/min": 0.2647233009338379, "sampling/importance_sampling_ratio/mean": 1.047873854637146, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3103306144475937, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.16511577926576138, "clip_ratio/high_max": 0.16511577926576138, "clip_ratio/region_mean": 0.16511577926576138, "reward_total_mean": 0.7880845665931702, "reward_meter_mean": 0.9107831716537476, "reward_meter_std": 0.10744628310203552, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9722753763198853, "reward_repeat_soft_std": 0.03736523538827896, "reward_judge_quality_mean": 0.6899999976158142, "reward_judge_quality_std": 0.28400203585624695, "reward_total_composite_mean": 0.7880845665931702, "reward_total_composite_std": 0.32319119572639465} {"timestamp_utc": "2026-04-12T23:59:11Z", "mode": "train", "global_step": 728, "epoch": 0.07312908086388749, "loss": -0.0521, "grad_norm": 5.275055885314941, "learning_rate": 7.796969696969697e-06, "num_tokens": 1345094.0, "completions/mean_length": 110.625, "completions/min_length": 31.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 53.28571701049805, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.42670655250549316, "rewards/meter/std": 0.3858836889266968, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.17817415297031403, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9625643491744995, "rewards/repeat_soft/std": 0.060829419642686844, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.13845446705818176, "rewards/total_composite/mean": 0.4607347548007965, "rewards/total_composite/std": 0.26191356778144836, "reward": 0.4607347548007965, "reward_std": 0.26191356778144836, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16681304574012756, "sampling/sampling_logp_difference/max": 1.4463651180267334, "sampling/importance_sampling_ratio/min": 0.2354244738817215, "sampling/importance_sampling_ratio/mean": 1.0476715564727783, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3929647654294968, "clip_ratio/low_mean": 0.055163171142339706, "clip_ratio/low_min": 0.055163171142339706, "clip_ratio/high_mean": 0.0964941754937172, "clip_ratio/high_max": 0.0964941754937172, "clip_ratio/region_mean": 0.1516573466360569, "reward_total_mean": 0.4607347548007965, "reward_meter_mean": 0.42670655250549316, "reward_meter_std": 0.3858836889266968, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.17817415297031403, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9625643491744995, "reward_repeat_soft_std": 0.060829419642686844, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.13845446705818176, "reward_total_composite_mean": 0.4607347548007965, "reward_total_composite_std": 0.26191356778144836} {"timestamp_utc": "2026-04-12T23:59:23Z", "mode": "train", "global_step": 729, "epoch": 0.07322953289804118, "loss": -0.0798, "grad_norm": 2.1341824531555176, "learning_rate": 7.793939393939394e-06, "num_tokens": 1346439.0, "completions/mean_length": 280.125, "completions/min_length": 44.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.5, "completions/mean_terminated_length": 48.25, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.5102807283401489, "rewards/meter/std": 0.423393577337265, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/hard_gate/mean": 0.5, "rewards/hard_gate/std": 0.5345224738121033, "rewards/repeat_soft/mean": 0.998033881187439, "rewards/repeat_soft/std": 0.0026844146195799112, "rewards/judge_quality/mean": 0.25874999165534973, "rewards/judge_quality/std": 0.2057347446680069, "rewards/total_composite/mean": 0.3552209138870239, "rewards/total_composite/std": 0.39364317059516907, "reward": 0.3552209138870239, "reward_std": 0.39364317059516907, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1680143028497696, "sampling/sampling_logp_difference/max": 1.2150444984436035, "sampling/importance_sampling_ratio/min": 0.29669681191444397, "sampling/importance_sampling_ratio/mean": 1.0487852096557617, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.678873673081398, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08643635688349605, "clip_ratio/high_max": 0.08643635688349605, "clip_ratio/region_mean": 0.08643635688349605, "reward_total_mean": 0.3552209138870239, "reward_meter_mean": 0.5102807283401489, "reward_meter_std": 0.423393577337265, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_hard_gate_mean": 0.5, "reward_hard_gate_std": 0.5345224738121033, "reward_repeat_soft_mean": 0.998033881187439, "reward_repeat_soft_std": 0.0026844146195799112, "reward_judge_quality_mean": 0.25874999165534973, "reward_judge_quality_std": 0.2057347446680069, "reward_total_composite_mean": 0.3552209138870239, "reward_total_composite_std": 0.39364317059516907} {"timestamp_utc": "2026-04-12T23:59:30Z", "mode": "train", "global_step": 730, "epoch": 0.07332998493219488, "loss": -0.0161, "grad_norm": 15.911628723144531, "learning_rate": 7.790909090909092e-06, "num_tokens": 1348530.0, "completions/mean_length": 62.375, "completions/min_length": 52.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.375, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.5229588150978088, "rewards/meter/std": 0.27087220549583435, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9843965768814087, "rewards/repeat_soft/std": 0.017212582752108574, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.6322711110115051, "rewards/total_composite/std": 0.13787156343460083, "reward": 0.6322711110115051, "reward_std": 0.13787154853343964, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18463730812072754, "sampling/sampling_logp_difference/max": 2.6229400634765625, "sampling/importance_sampling_ratio/min": 0.07258912920951843, "sampling/importance_sampling_ratio/mean": 1.017288327217102, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5111111104488373, "clip_ratio/low_mean": 0.08806340768933296, "clip_ratio/low_min": 0.08806340768933296, "clip_ratio/high_mean": 0.05992140807211399, "clip_ratio/high_max": 0.05992140807211399, "clip_ratio/region_mean": 0.14798481576144695, "reward_total_mean": 0.6322711110115051, "reward_meter_mean": 0.5229588150978088, "reward_meter_std": 0.27087220549583435, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9843965768814087, "reward_repeat_soft_std": 0.017212582752108574, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.6322711110115051, "reward_total_composite_std": 0.13787156343460083} {"timestamp_utc": "2026-04-12T23:59:37Z", "mode": "train", "global_step": 731, "epoch": 0.07343043696634857, "loss": 0.0767, "grad_norm": 13.141799926757812, "learning_rate": 7.787878787878789e-06, "num_tokens": 1350432.0, "completions/mean_length": 68.75, "completions/min_length": 58.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.75, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.5356042385101318, "rewards/meter/std": 0.36468902230262756, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9901726245880127, "rewards/repeat_soft/std": 0.008687156252563, "rewards/judge_quality/mean": 0.4024999737739563, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.5826641321182251, "rewards/total_composite/std": 0.145517036318779, "reward": 0.5826641321182251, "reward_std": 0.1455170214176178, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20888997614383698, "sampling/sampling_logp_difference/max": 1.2988157272338867, "sampling/importance_sampling_ratio/min": 0.272854745388031, "sampling/importance_sampling_ratio/mean": 1.0503722429275513, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2472992837429047, "clip_ratio/low_mean": 0.10841447860002518, "clip_ratio/low_min": 0.10841447860002518, "clip_ratio/high_mean": 0.07929808087646961, "clip_ratio/high_max": 0.07929808087646961, "clip_ratio/region_mean": 0.1877125594764948, "reward_total_mean": 0.5826641321182251, "reward_meter_mean": 0.5356042385101318, "reward_meter_std": 0.36468902230262756, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9901726245880127, "reward_repeat_soft_std": 0.008687156252563, "reward_judge_quality_mean": 0.4024999737739563, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.5826641321182251, "reward_total_composite_std": 0.145517036318779} {"timestamp_utc": "2026-04-12T23:59:44Z", "mode": "train", "global_step": 732, "epoch": 0.07353088900050227, "loss": -0.0282, "grad_norm": 9.167341232299805, "learning_rate": 7.784848484848484e-06, "num_tokens": 1352636.0, "completions/mean_length": 96.5, "completions/min_length": 74.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.5, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.9765560626983643, "rewards/meter/std": 0.034577954560518265, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9447321891784668, "rewards/repeat_soft/std": 0.09779856353998184, "rewards/judge_quality/mean": 0.6649999618530273, "rewards/judge_quality/std": 0.2622430920600891, "rewards/total_composite/mean": 0.8534234166145325, "rewards/total_composite/std": 0.06874643266201019, "reward": 0.8534234166145325, "reward_std": 0.068746417760849, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1712886542081833, "sampling/sampling_logp_difference/max": 1.7384939193725586, "sampling/importance_sampling_ratio/min": 0.17578494548797607, "sampling/importance_sampling_ratio/mean": 1.0195978879928589, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6457203850150108, "clip_ratio/low_mean": 0.10752711817622185, "clip_ratio/low_min": 0.10752711817622185, "clip_ratio/high_mean": 0.07831738796085119, "clip_ratio/high_max": 0.07831738796085119, "clip_ratio/region_mean": 0.18584450613707304, "reward_total_mean": 0.8534234166145325, "reward_meter_mean": 0.9765560626983643, "reward_meter_std": 0.034577954560518265, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9447321891784668, "reward_repeat_soft_std": 0.09779856353998184, "reward_judge_quality_mean": 0.6649999618530273, "reward_judge_quality_std": 0.2622430920600891, "reward_total_composite_mean": 0.8534234166145325, "reward_total_composite_std": 0.06874643266201019} {"timestamp_utc": "2026-04-12T23:59:50Z", "mode": "train", "global_step": 733, "epoch": 0.07363134103465595, "loss": 0.0396, "grad_norm": 8.228132247924805, "learning_rate": 7.781818181818183e-06, "num_tokens": 1354040.0, "completions/mean_length": 33.5, "completions/min_length": 31.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.5, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.95853590965271, "rewards/meter/std": 0.0851975753903389, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 1.0, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.9200000166893005, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.957341194152832, "rewards/total_composite/std": 0.038338907063007355, "reward": 0.957341194152832, "reward_std": 0.038338903337717056, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0932590514421463, "sampling/sampling_logp_difference/max": 2.638167381286621, "sampling/importance_sampling_ratio/min": 0.07149216532707214, "sampling/importance_sampling_ratio/mean": 1.0204534530639648, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.41290615499019623, "clip_ratio/low_mean": 0.0069444444961845875, "clip_ratio/low_min": 0.0069444444961845875, "clip_ratio/high_mean": 0.07205594936385751, "clip_ratio/high_max": 0.07205594936385751, "clip_ratio/region_mean": 0.0790003938600421, "reward_total_mean": 0.957341194152832, "reward_meter_mean": 0.95853590965271, "reward_meter_std": 0.0851975753903389, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 1.0, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.9200000166893005, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.957341194152832, "reward_total_composite_std": 0.038338907063007355} {"timestamp_utc": "2026-04-13T00:00:03Z", "mode": "train", "global_step": 734, "epoch": 0.07373179306880964, "loss": -0.0666, "grad_norm": 1.7756036520004272, "learning_rate": 7.778787878787879e-06, "num_tokens": 1355551.0, "completions/mean_length": 214.875, "completions/min_length": 33.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 36.60000228881836, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.17558440566062927, "rewards/meter/std": 0.3385484218597412, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9940953254699707, "rewards/repeat_soft/std": 0.012902759946882725, "rewards/judge_quality/mean": 0.413750022649765, "rewards/judge_quality/std": 0.3609882593154907, "rewards/total_composite/mean": 0.35329777002334595, "rewards/total_composite/std": 0.3429962694644928, "reward": 0.35329777002334595, "reward_std": 0.3429962396621704, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16232700645923615, "sampling/sampling_logp_difference/max": 1.054856300354004, "sampling/importance_sampling_ratio/min": 0.3482424318790436, "sampling/importance_sampling_ratio/mean": 1.062116026878357, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.969492644071579, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08141645370051265, "clip_ratio/high_max": 0.08141645370051265, "clip_ratio/region_mean": 0.08141645370051265, "reward_total_mean": 0.35329777002334595, "reward_meter_mean": 0.17558440566062927, "reward_meter_std": 0.3385484218597412, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9940953254699707, "reward_repeat_soft_std": 0.012902759946882725, "reward_judge_quality_mean": 0.413750022649765, "reward_judge_quality_std": 0.3609882593154907, "reward_total_composite_mean": 0.35329777002334595, "reward_total_composite_std": 0.3429962694644928} {"timestamp_utc": "2026-04-13T00:00:10Z", "mode": "train", "global_step": 735, "epoch": 0.07383224510296334, "loss": -0.0201, "grad_norm": 20.62511444091797, "learning_rate": 7.775757575757576e-06, "num_tokens": 1357038.0, "completions/mean_length": 23.875, "completions/min_length": 18.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.875, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.6196253299713135, "rewards/meter/std": 0.4184964597225189, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.6480814218521118, "rewards/total_composite/std": 0.18281583487987518, "reward": 0.6480814218521118, "reward_std": 0.18281583487987518, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22678668797016144, "sampling/sampling_logp_difference/max": 1.8481054306030273, "sampling/importance_sampling_ratio/min": 0.1575353443622589, "sampling/importance_sampling_ratio/mean": 1.061033844947815, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0164593309164047, "clip_ratio/low_mean": 0.05440789461135864, "clip_ratio/low_min": 0.05440789461135864, "clip_ratio/high_mean": 0.11500000301748514, "clip_ratio/high_max": 0.11500000301748514, "clip_ratio/region_mean": 0.16940789762884378, "reward_total_mean": 0.6480814218521118, "reward_meter_mean": 0.6196253299713135, "reward_meter_std": 0.4184964597225189, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.6480814218521118, "reward_total_composite_std": 0.18281583487987518} {"timestamp_utc": "2026-04-13T00:00:24Z", "mode": "train", "global_step": 736, "epoch": 0.07393269713711703, "loss": -0.1092, "grad_norm": 1.7332168817520142, "learning_rate": 7.772727272727273e-06, "num_tokens": 1358618.0, "completions/mean_length": 225.5, "completions/min_length": 50.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 53.60000228881836, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.691292405128479, "rewards/meter/std": 0.35428351163864136, "rewards/count_adherence/mean": 0.71875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9945201873779297, "rewards/repeat_soft/std": 0.007023094221949577, "rewards/judge_quality/mean": 0.25999999046325684, "rewards/judge_quality/std": 0.18314708769321442, "rewards/total_composite/mean": 0.46993470191955566, "rewards/total_composite/std": 0.3899978995323181, "reward": 0.46993470191955566, "reward_std": 0.3899978995323181, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18782144784927368, "sampling/sampling_logp_difference/max": 1.9945554733276367, "sampling/importance_sampling_ratio/min": 0.13607412576675415, "sampling/importance_sampling_ratio/mean": 1.0604511499404907, "sampling/importance_sampling_ratio/max": 1.9542254209518433, "entropy": 1.3344504088163376, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.112009447067976, "clip_ratio/high_max": 0.112009447067976, "clip_ratio/region_mean": 0.112009447067976, "reward_total_mean": 0.46993470191955566, "reward_meter_mean": 0.691292405128479, "reward_meter_std": 0.35428351163864136, "reward_count_adherence_mean": 0.71875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9945201873779297, "reward_repeat_soft_std": 0.007023094221949577, "reward_judge_quality_mean": 0.25999999046325684, "reward_judge_quality_std": 0.18314708769321442, "reward_total_composite_mean": 0.46993470191955566, "reward_total_composite_std": 0.3899978995323181} {"timestamp_utc": "2026-04-13T00:00:33Z", "mode": "train", "global_step": 737, "epoch": 0.07403314917127071, "loss": 0.0435, "grad_norm": 19.795791625976562, "learning_rate": 7.76969696969697e-06, "num_tokens": 1359948.0, "completions/mean_length": 24.25, "completions/min_length": 19.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.25, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.8213329315185547, "rewards/meter/std": 0.25081706047058105, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9602519273757935, "rewards/repeat_soft/std": 0.006358357612043619, "rewards/judge_quality/mean": 0.5087499618530273, "rewards/judge_quality/std": 0.16617010533809662, "rewards/total_composite/mean": 0.7682499885559082, "rewards/total_composite/std": 0.13551296293735504, "reward": 0.7682499885559082, "reward_std": 0.13551299273967743, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1930420845746994, "sampling/sampling_logp_difference/max": 2.3575363159179688, "sampling/importance_sampling_ratio/min": 0.09465312957763672, "sampling/importance_sampling_ratio/mean": 1.0101193189620972, "sampling/importance_sampling_ratio/max": 1.7054084539413452, "entropy": 1.687011569738388, "clip_ratio/low_mean": 0.04902087710797787, "clip_ratio/low_min": 0.04902087710797787, "clip_ratio/high_mean": 0.08017241396009922, "clip_ratio/high_max": 0.08017241396009922, "clip_ratio/region_mean": 0.1291932910680771, "reward_total_mean": 0.7682499885559082, "reward_meter_mean": 0.8213329315185547, "reward_meter_std": 0.25081706047058105, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9602519273757935, "reward_repeat_soft_std": 0.006358357612043619, "reward_judge_quality_mean": 0.5087499618530273, "reward_judge_quality_std": 0.16617010533809662, "reward_total_composite_mean": 0.7682499885559082, "reward_total_composite_std": 0.13551296293735504} {"timestamp_utc": "2026-04-13T00:00:43Z", "mode": "train", "global_step": 738, "epoch": 0.07413360120542441, "loss": 0.0342, "grad_norm": 12.235231399536133, "learning_rate": 7.766666666666666e-06, "num_tokens": 1361692.0, "completions/mean_length": 40.0, "completions/min_length": 34.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.42330488562583923, "rewards/meter/std": 0.3256361186504364, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9493792057037354, "rewards/repeat_soft/std": 0.042173273861408234, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.25587037205696106, "rewards/total_composite/mean": 0.618800163269043, "rewards/total_composite/std": 0.10031331330537796, "reward": 0.618800163269043, "reward_std": 0.10031331330537796, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14092423021793365, "sampling/sampling_logp_difference/max": 1.7978935241699219, "sampling/importance_sampling_ratio/min": 0.1656474471092224, "sampling/importance_sampling_ratio/mean": 1.02724289894104, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1048939004540443, "clip_ratio/low_mean": 0.09005310665816069, "clip_ratio/low_min": 0.09005310665816069, "clip_ratio/high_mean": 0.04811508068814874, "clip_ratio/high_max": 0.04811508068814874, "clip_ratio/region_mean": 0.13816818734630942, "reward_total_mean": 0.618800163269043, "reward_meter_mean": 0.42330488562583923, "reward_meter_std": 0.3256361186504364, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9493792057037354, "reward_repeat_soft_std": 0.042173273861408234, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.25587037205696106, "reward_total_composite_mean": 0.618800163269043, "reward_total_composite_std": 0.10031331330537796} {"timestamp_utc": "2026-04-13T00:00:52Z", "mode": "train", "global_step": 739, "epoch": 0.0742340532395781, "loss": 0.0998, "grad_norm": 13.304032325744629, "learning_rate": 7.763636363636364e-06, "num_tokens": 1363664.0, "completions/mean_length": 69.5, "completions/min_length": 59.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.5, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.3993501663208008, "rewards/meter/std": 0.36814042925834656, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9909764528274536, "rewards/repeat_soft/std": 0.011509276926517487, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.5473052263259888, "rewards/total_composite/std": 0.15826629102230072, "reward": 0.5473052263259888, "reward_std": 0.15826627612113953, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21991553902626038, "sampling/sampling_logp_difference/max": 2.1066808700561523, "sampling/importance_sampling_ratio/min": 0.12164103984832764, "sampling/importance_sampling_ratio/mean": 1.0144729614257812, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2064418643712997, "clip_ratio/low_mean": 0.09213462192565203, "clip_ratio/low_min": 0.09213462192565203, "clip_ratio/high_mean": 0.08875676803290844, "clip_ratio/high_max": 0.08875676803290844, "clip_ratio/region_mean": 0.18089138995856047, "reward_total_mean": 0.5473052263259888, "reward_meter_mean": 0.3993501663208008, "reward_meter_std": 0.36814042925834656, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9909764528274536, "reward_repeat_soft_std": 0.011509276926517487, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.5473052263259888, "reward_total_composite_std": 0.15826629102230072} {"timestamp_utc": "2026-04-13T00:01:12Z", "mode": "train", "global_step": 740, "epoch": 0.0743345052737318, "loss": -0.1613, "grad_norm": 11.659581184387207, "learning_rate": 7.76060606060606e-06, "num_tokens": 1365031.0, "completions/mean_length": 26.875, "completions/min_length": 15.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.875, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.780845046043396, "rewards/meter/std": 0.347287118434906, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.8008802533149719, "rewards/total_composite/std": 0.18332539498806, "reward": 0.8008802533149719, "reward_std": 0.18332539498806, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13185633718967438, "sampling/sampling_logp_difference/max": 1.1602792739868164, "sampling/importance_sampling_ratio/min": 0.31339865922927856, "sampling/importance_sampling_ratio/mean": 1.0224130153656006, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1099969744682312, "clip_ratio/low_mean": 0.09697580523788929, "clip_ratio/low_min": 0.09697580523788929, "clip_ratio/high_mean": 0.09991707094013691, "clip_ratio/high_max": 0.09991707094013691, "clip_ratio/region_mean": 0.1968928761780262, "reward_total_mean": 0.8008802533149719, "reward_meter_mean": 0.780845046043396, "reward_meter_std": 0.347287118434906, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.8008802533149719, "reward_total_composite_std": 0.18332539498806} {"timestamp_utc": "2026-04-13T00:01:20Z", "mode": "train", "global_step": 741, "epoch": 0.07443495730788549, "loss": 0.0286, "grad_norm": 14.41140365600586, "learning_rate": 7.757575757575758e-06, "num_tokens": 1366683.0, "completions/mean_length": 37.5, "completions/min_length": 30.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.5, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.7935720086097717, "rewards/meter/std": 0.31199753284454346, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9990912675857544, "rewards/repeat_soft/std": 0.0015775051433593035, "rewards/judge_quality/mean": 0.4437499940395355, "rewards/judge_quality/std": 0.13845448195934296, "rewards/total_composite/mean": 0.7401415109634399, "rewards/total_composite/std": 0.12343714386224747, "reward": 0.7401415109634399, "reward_std": 0.12343714386224747, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2020515650510788, "sampling/sampling_logp_difference/max": 1.2219257354736328, "sampling/importance_sampling_ratio/min": 0.2946621775627136, "sampling/importance_sampling_ratio/mean": 1.0120500326156616, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.061838060617447, "clip_ratio/low_mean": 0.05985577031970024, "clip_ratio/low_min": 0.05985577031970024, "clip_ratio/high_mean": 0.12528770044445992, "clip_ratio/high_max": 0.12528770044445992, "clip_ratio/region_mean": 0.18514347076416016, "reward_total_mean": 0.7401415109634399, "reward_meter_mean": 0.7935720086097717, "reward_meter_std": 0.31199753284454346, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9990912675857544, "reward_repeat_soft_std": 0.0015775051433593035, "reward_judge_quality_mean": 0.4437499940395355, "reward_judge_quality_std": 0.13845448195934296, "reward_total_composite_mean": 0.7401415109634399, "reward_total_composite_std": 0.12343714386224747} {"timestamp_utc": "2026-04-13T00:01:34Z", "mode": "train", "global_step": 742, "epoch": 0.07453540934203917, "loss": 0.0028, "grad_norm": 13.610276222229004, "learning_rate": 7.754545454545455e-06, "num_tokens": 1368460.0, "completions/mean_length": 46.125, "completions/min_length": 40.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.125, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.9469984769821167, "rewards/meter/std": 0.10346122086048126, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9923200011253357, "rewards/repeat_soft/std": 0.013966222293674946, "rewards/judge_quality/mean": 0.5362499952316284, "rewards/judge_quality/std": 0.22385822236537933, "rewards/total_composite/mean": 0.8362562656402588, "rewards/total_composite/std": 0.0974682942032814, "reward": 0.8362562656402588, "reward_std": 0.0974682867527008, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14064371585845947, "sampling/sampling_logp_difference/max": 1.4551105499267578, "sampling/importance_sampling_ratio/min": 0.23337455093860626, "sampling/importance_sampling_ratio/mean": 1.0284279584884644, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4242420867085457, "clip_ratio/low_mean": 0.0726204663515091, "clip_ratio/low_min": 0.0726204663515091, "clip_ratio/high_mean": 0.04411139478906989, "clip_ratio/high_max": 0.04411139478906989, "clip_ratio/region_mean": 0.11673186114057899, "reward_total_mean": 0.8362562656402588, "reward_meter_mean": 0.9469984769821167, "reward_meter_std": 0.10346122086048126, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9923200011253357, "reward_repeat_soft_std": 0.013966222293674946, "reward_judge_quality_mean": 0.5362499952316284, "reward_judge_quality_std": 0.22385822236537933, "reward_total_composite_mean": 0.8362562656402588, "reward_total_composite_std": 0.0974682942032814} {"timestamp_utc": "2026-04-13T00:01:49Z", "mode": "train", "global_step": 743, "epoch": 0.07463586137619287, "loss": -0.158, "grad_norm": 2.3502397537231445, "learning_rate": 7.751515151515153e-06, "num_tokens": 1370281.0, "completions/mean_length": 118.625, "completions/min_length": 41.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 62.42857360839844, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.9707833528518677, "rewards/meter/std": 0.024299966171383858, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9902569651603699, "rewards/repeat_soft/std": 0.018241362646222115, "rewards/judge_quality/mean": 0.39375001192092896, "rewards/judge_quality/std": 0.18943053483963013, "rewards/total_composite/mean": 0.6878981590270996, "rewards/total_composite/std": 0.2800547778606415, "reward": 0.6878981590270996, "reward_std": 0.2800547778606415, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18938440084457397, "sampling/sampling_logp_difference/max": 1.5141520500183105, "sampling/importance_sampling_ratio/min": 0.2199946492910385, "sampling/importance_sampling_ratio/mean": 1.040883183479309, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8332430571317673, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.18024134635925293, "clip_ratio/high_max": 0.18024134635925293, "clip_ratio/region_mean": 0.18024134635925293, "reward_total_mean": 0.6878981590270996, "reward_meter_mean": 0.9707833528518677, "reward_meter_std": 0.024299966171383858, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9902569651603699, "reward_repeat_soft_std": 0.018241362646222115, "reward_judge_quality_mean": 0.39375001192092896, "reward_judge_quality_std": 0.18943053483963013, "reward_total_composite_mean": 0.6878981590270996, "reward_total_composite_std": 0.2800547778606415} {"timestamp_utc": "2026-04-13T00:02:02Z", "mode": "train", "global_step": 744, "epoch": 0.07473631341034656, "loss": -0.1756, "grad_norm": 3.398122549057007, "learning_rate": 7.74848484848485e-06, "num_tokens": 1372262.0, "completions/mean_length": 144.625, "completions/min_length": 81.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 92.14286041259766, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.7656146287918091, "rewards/meter/std": 0.33805158734321594, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.951854944229126, "rewards/repeat_soft/std": 0.04391444846987724, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.18845234811306, "rewards/total_composite/mean": 0.6206774711608887, "rewards/total_composite/std": 0.28685200214385986, "reward": 0.6206774711608887, "reward_std": 0.28685200214385986, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16546472907066345, "sampling/sampling_logp_difference/max": 1.6801092624664307, "sampling/importance_sampling_ratio/min": 0.18635360896587372, "sampling/importance_sampling_ratio/mean": 1.0247094631195068, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4579652398824692, "clip_ratio/low_mean": 0.022590361535549164, "clip_ratio/low_min": 0.022590361535549164, "clip_ratio/high_mean": 0.138014055788517, "clip_ratio/high_max": 0.138014055788517, "clip_ratio/region_mean": 0.16060441732406616, "reward_total_mean": 0.6206774711608887, "reward_meter_mean": 0.7656146287918091, "reward_meter_std": 0.33805158734321594, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.951854944229126, "reward_repeat_soft_std": 0.04391444846987724, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.18845234811306, "reward_total_composite_mean": 0.6206774711608887, "reward_total_composite_std": 0.28685200214385986} {"timestamp_utc": "2026-04-13T00:02:11Z", "mode": "train", "global_step": 745, "epoch": 0.07483676544450026, "loss": 0.0615, "grad_norm": 9.528594970703125, "learning_rate": 7.745454545454545e-06, "num_tokens": 1373953.0, "completions/mean_length": 59.375, "completions/min_length": 50.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.375, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9926562309265137, "rewards/meter/std": 0.004592522047460079, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9235355854034424, "rewards/repeat_soft/std": 0.1248006746172905, "rewards/judge_quality/mean": 0.44749999046325684, "rewards/judge_quality/std": 0.30555570125579834, "rewards/total_composite/mean": 0.8232988119125366, "rewards/total_composite/std": 0.0991811528801918, "reward": 0.8232988119125366, "reward_std": 0.0991811603307724, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17104260623455048, "sampling/sampling_logp_difference/max": 1.5248444080352783, "sampling/importance_sampling_ratio/min": 0.21765492856502533, "sampling/importance_sampling_ratio/mean": 1.0407902002334595, "sampling/importance_sampling_ratio/max": 1.9790178537368774, "entropy": 1.9034627676010132, "clip_ratio/low_mean": 0.13073908351361752, "clip_ratio/low_min": 0.13073908351361752, "clip_ratio/high_mean": 0.035624999552965164, "clip_ratio/high_max": 0.035624999552965164, "clip_ratio/region_mean": 0.16636408306658268, "reward_total_mean": 0.8232988119125366, "reward_meter_mean": 0.9926562309265137, "reward_meter_std": 0.004592522047460079, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9235355854034424, "reward_repeat_soft_std": 0.1248006746172905, "reward_judge_quality_mean": 0.44749999046325684, "reward_judge_quality_std": 0.30555570125579834, "reward_total_composite_mean": 0.8232988119125366, "reward_total_composite_std": 0.0991811528801918} {"timestamp_utc": "2026-04-13T00:02:25Z", "mode": "train", "global_step": 746, "epoch": 0.07493721747865394, "loss": -0.084, "grad_norm": 1.4537217617034912, "learning_rate": 7.742424242424244e-06, "num_tokens": 1375437.0, "completions/mean_length": 223.5, "completions/min_length": 48.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 50.400001525878906, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.7046393156051636, "rewards/meter/std": 0.34302273392677307, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9946229457855225, "rewards/repeat_soft/std": 0.0062688072212040424, "rewards/judge_quality/mean": 0.26374998688697815, "rewards/judge_quality/std": 0.18715444207191467, "rewards/total_composite/mean": 0.4519021213054657, "rewards/total_composite/std": 0.3225303590297699, "reward": 0.4519021213054657, "reward_std": 0.3225303292274475, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1703650802373886, "sampling/sampling_logp_difference/max": 1.234588623046875, "sampling/importance_sampling_ratio/min": 0.29095444083213806, "sampling/importance_sampling_ratio/mean": 1.0367274284362793, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.891614243388176, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09041414270177484, "clip_ratio/high_max": 0.09041414270177484, "clip_ratio/region_mean": 0.09041414270177484, "reward_total_mean": 0.4519021213054657, "reward_meter_mean": 0.7046393156051636, "reward_meter_std": 0.34302273392677307, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9946229457855225, "reward_repeat_soft_std": 0.0062688072212040424, "reward_judge_quality_mean": 0.26374998688697815, "reward_judge_quality_std": 0.18715444207191467, "reward_total_composite_mean": 0.4519021213054657, "reward_total_composite_std": 0.3225303590297699} {"timestamp_utc": "2026-04-13T00:02:34Z", "mode": "train", "global_step": 747, "epoch": 0.07503766951280763, "loss": 0.0109, "grad_norm": 20.17443084716797, "learning_rate": 7.73939393939394e-06, "num_tokens": 1376835.0, "completions/mean_length": 20.75, "completions/min_length": 17.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.75, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.31436869502067566, "rewards/meter/std": 0.4122677743434906, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.878701388835907, "rewards/repeat_soft/std": 0.08967190235853195, "rewards/judge_quality/mean": 0.39750000834465027, "rewards/judge_quality/std": 0.10110107809305191, "rewards/total_composite/mean": 0.4985860586166382, "rewards/total_composite/std": 0.1644967943429947, "reward": 0.4985860586166382, "reward_std": 0.1644967943429947, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1471824198961258, "sampling/sampling_logp_difference/max": 1.2816619873046875, "sampling/importance_sampling_ratio/min": 0.2775755822658539, "sampling/importance_sampling_ratio/mean": 0.9790687561035156, "sampling/importance_sampling_ratio/max": 1.863194465637207, "entropy": 1.0636684224009514, "clip_ratio/low_mean": 0.12016262486577034, "clip_ratio/low_min": 0.12016262486577034, "clip_ratio/high_mean": 0.057894738391041756, "clip_ratio/high_max": 0.057894738391041756, "clip_ratio/region_mean": 0.1780573632568121, "reward_total_mean": 0.4985860586166382, "reward_meter_mean": 0.31436869502067566, "reward_meter_std": 0.4122677743434906, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.878701388835907, "reward_repeat_soft_std": 0.08967190235853195, "reward_judge_quality_mean": 0.39750000834465027, "reward_judge_quality_std": 0.10110107809305191, "reward_total_composite_mean": 0.4985860586166382, "reward_total_composite_std": 0.1644967943429947} {"timestamp_utc": "2026-04-13T00:02:42Z", "mode": "train", "global_step": 748, "epoch": 0.07513812154696133, "loss": 0.0025, "grad_norm": 13.700549125671387, "learning_rate": 7.736363636363637e-06, "num_tokens": 1378494.0, "completions/mean_length": 51.375, "completions/min_length": 32.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.375, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9646717309951782, "rewards/meter/std": 0.06012219563126564, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9542040824890137, "rewards/repeat_soft/std": 0.044572919607162476, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8066476583480835, "rewards/total_composite/std": 0.027215462177991867, "reward": 0.8066476583480835, "reward_std": 0.027215473353862762, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15016280114650726, "sampling/sampling_logp_difference/max": 1.457235336303711, "sampling/importance_sampling_ratio/min": 0.23287922143936157, "sampling/importance_sampling_ratio/mean": 1.0324002504348755, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1535938680171967, "clip_ratio/low_mean": 0.025607638992369175, "clip_ratio/low_min": 0.025607638992369175, "clip_ratio/high_mean": 0.10378080792725086, "clip_ratio/high_max": 0.10378080792725086, "clip_ratio/region_mean": 0.12938844691962004, "reward_total_mean": 0.8066476583480835, "reward_meter_mean": 0.9646717309951782, "reward_meter_std": 0.06012219563126564, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9542040824890137, "reward_repeat_soft_std": 0.044572919607162476, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8066476583480835, "reward_total_composite_std": 0.027215462177991867} {"timestamp_utc": "2026-04-13T00:02:49Z", "mode": "train", "global_step": 749, "epoch": 0.07523857358111502, "loss": 0.1655, "grad_norm": 16.65985870361328, "learning_rate": 7.733333333333334e-06, "num_tokens": 1380316.0, "completions/mean_length": 52.75, "completions/min_length": 40.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.75, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9310469627380371, "rewards/meter/std": 0.0944700688123703, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.15430334210395813, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9700033664703369, "rewards/repeat_soft/std": 0.024646032601594925, "rewards/judge_quality/mean": 0.4024999737739563, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.7492214441299438, "rewards/total_composite/std": 0.04906798154115677, "reward": 0.7492214441299438, "reward_std": 0.04906797781586647, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14224547147750854, "sampling/sampling_logp_difference/max": 1.0782742500305176, "sampling/importance_sampling_ratio/min": 0.3401820957660675, "sampling/importance_sampling_ratio/mean": 1.0298473834991455, "sampling/importance_sampling_ratio/max": 1.995485782623291, "entropy": 1.363947868347168, "clip_ratio/low_mean": 0.026668205857276917, "clip_ratio/low_min": 0.026668205857276917, "clip_ratio/high_mean": 0.1241611884906888, "clip_ratio/high_max": 0.1241611884906888, "clip_ratio/region_mean": 0.15082939434796572, "reward_total_mean": 0.7492214441299438, "reward_meter_mean": 0.9310469627380371, "reward_meter_std": 0.0944700688123703, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.15430334210395813, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9700033664703369, "reward_repeat_soft_std": 0.024646032601594925, "reward_judge_quality_mean": 0.4024999737739563, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.7492214441299438, "reward_total_composite_std": 0.04906798154115677} {"timestamp_utc": "2026-04-13T00:03:01Z", "mode": "train", "global_step": 750, "epoch": 0.07533902561526871, "loss": 0.0388, "grad_norm": 17.75530242919922, "learning_rate": 7.730303030303032e-06, "num_tokens": 1382135.0, "completions/mean_length": 32.375, "completions/min_length": 29.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.375, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.7801449298858643, "rewards/meter/std": 0.28773176670074463, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9651429653167725, "rewards/repeat_soft/std": 0.052871670573949814, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.2499571591615677, "rewards/total_composite/mean": 0.7558295130729675, "rewards/total_composite/std": 0.16796299815177917, "reward": 0.7558295130729675, "reward_std": 0.16796299815177917, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2088986188173294, "sampling/sampling_logp_difference/max": 1.632988452911377, "sampling/importance_sampling_ratio/min": 0.19534491002559662, "sampling/importance_sampling_ratio/mean": 1.0330525636672974, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3712529242038727, "clip_ratio/low_mean": 0.09933393076062202, "clip_ratio/low_min": 0.09933393076062202, "clip_ratio/high_mean": 0.08632914628833532, "clip_ratio/high_max": 0.08632914628833532, "clip_ratio/region_mean": 0.18566307704895735, "reward_total_mean": 0.7558295130729675, "reward_meter_mean": 0.7801449298858643, "reward_meter_std": 0.28773176670074463, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9651429653167725, "reward_repeat_soft_std": 0.052871670573949814, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.2499571591615677, "reward_total_composite_mean": 0.7558295130729675, "reward_total_composite_std": 0.16796299815177917} {"timestamp_utc": "2026-04-13T00:04:20Z", "mode": "eval", "global_step": 750, "epoch": 0.07533902561526871, "eval_loss": NaN, "eval_runtime": 79.4468, "eval_samples_per_second": 1.007, "eval_steps_per_second": 0.126, "eval_num_tokens": 1382135.0, "eval_completions/mean_length": 85.6375, "eval_completions/min_length": 29.0, "eval_completions/max_length": 228.5, "eval_completions/clipped_ratio": 0.05, "eval_completions/mean_terminated_length": 63.735714721679685, "eval_completions/min_terminated_length": 29.0, "eval_completions/max_terminated_length": 105.2, "eval_rewards/meter/mean": 0.7649060547351837, "eval_rewards/meter/std": 0.3034533068537712, "eval_rewards/count_adherence/mean": 0.8691666543483734, "eval_rewards/count_adherence/std": 0.13377416357398034, "eval_rewards/hard_gate/mean": 0.925, "eval_rewards/hard_gate/std": 0.1632926881313324, "eval_rewards/repeat_soft/mean": 0.968361747264862, "eval_rewards/repeat_soft/std": 0.0456226734444499, "eval_rewards/judge_quality/mean": 0.4598749965429306, "eval_rewards/judge_quality/std": 0.18980545550584793, "eval_rewards/total_composite/mean": 0.672697389125824, "eval_rewards/total_composite/std": 0.2052470114082098, "eval_reward": 0.672697389125824, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.10140573754906654, "eval_sampling/sampling_logp_difference/max": 1.07471923828125, "eval_sampling/importance_sampling_ratio/min": 0.34526675641536714, "eval_sampling/importance_sampling_ratio/mean": 1.0281419038772583, "eval_sampling/importance_sampling_ratio/max": 1.5135164976119995, "eval_entropy": 1.2778119087219237, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.672697389125824, "eval_reward_meter_mean": 0.7649060547351837, "eval_reward_meter_std": 0.3034533068537712, "eval_reward_count_adherence_mean": 0.8691666543483734, "eval_reward_count_adherence_std": 0.13377416357398034, "eval_reward_hard_gate_mean": 0.925, "eval_reward_hard_gate_std": 0.1632926881313324, "eval_reward_repeat_soft_mean": 0.968361747264862, "eval_reward_repeat_soft_std": 0.0456226734444499, "eval_reward_judge_quality_mean": 0.4598749965429306, "eval_reward_judge_quality_std": 0.18980545550584793, "eval_reward_total_composite_mean": 0.672697389125824, "eval_reward_total_composite_std": 0.2052470114082098} {"timestamp_utc": "2026-04-13T00:04:38Z", "mode": "train", "global_step": 751, "epoch": 0.0754394776494224, "loss": -0.1086, "grad_norm": 1.813501000404358, "learning_rate": 7.727272727272727e-06, "num_tokens": 1383673.0, "completions/mean_length": 101.25, "completions/min_length": 31.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 42.57143020629883, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.8358834385871887, "rewards/meter/std": 0.3160637319087982, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.30860671401023865, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9919929504394531, "rewards/repeat_soft/std": 0.008503753691911697, "rewards/judge_quality/mean": 0.5512499809265137, "rewards/judge_quality/std": 0.30990493297576904, "rewards/total_composite/mean": 0.7407218217849731, "rewards/total_composite/std": 0.2517816126346588, "reward": 0.7407218217849731, "reward_std": 0.2517815828323364, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15586720407009125, "sampling/sampling_logp_difference/max": 1.236461877822876, "sampling/importance_sampling_ratio/min": 0.2904099225997925, "sampling/importance_sampling_ratio/mean": 1.030277132987976, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9710303917527199, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.11771242693066597, "clip_ratio/high_max": 0.11771242693066597, "clip_ratio/region_mean": 0.11771242693066597, "reward_total_mean": 0.7407218217849731, "reward_meter_mean": 0.8358834385871887, "reward_meter_std": 0.3160637319087982, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.30860671401023865, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9919929504394531, "reward_repeat_soft_std": 0.008503753691911697, "reward_judge_quality_mean": 0.5512499809265137, "reward_judge_quality_std": 0.30990493297576904, "reward_total_composite_mean": 0.7407218217849731, "reward_total_composite_std": 0.2517816126346588} {"timestamp_utc": "2026-04-13T00:04:47Z", "mode": "train", "global_step": 752, "epoch": 0.07553992968357609, "loss": 0.0189, "grad_norm": 12.216497421264648, "learning_rate": 7.724242424242424e-06, "num_tokens": 1385891.0, "completions/mean_length": 80.25, "completions/min_length": 59.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.25, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.8088330030441284, "rewards/meter/std": 0.33025532960891724, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9821277856826782, "rewards/repeat_soft/std": 0.009899999015033245, "rewards/judge_quality/mean": 0.5862500071525574, "rewards/judge_quality/std": 0.22984081506729126, "rewards/total_composite/mean": 0.6973103880882263, "rewards/total_composite/std": 0.2921626567840576, "reward": 0.6973103880882263, "reward_std": 0.2921626567840576, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1659153401851654, "sampling/sampling_logp_difference/max": 1.7635149955749512, "sampling/importance_sampling_ratio/min": 0.1714411824941635, "sampling/importance_sampling_ratio/mean": 1.0180912017822266, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4188986867666245, "clip_ratio/low_mean": 0.04219202138483524, "clip_ratio/low_min": 0.04219202138483524, "clip_ratio/high_mean": 0.11517438385635614, "clip_ratio/high_max": 0.11517438385635614, "clip_ratio/region_mean": 0.1573664052411914, "reward_total_mean": 0.6973103880882263, "reward_meter_mean": 0.8088330030441284, "reward_meter_std": 0.33025532960891724, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9821277856826782, "reward_repeat_soft_std": 0.009899999015033245, "reward_judge_quality_mean": 0.5862500071525574, "reward_judge_quality_std": 0.22984081506729126, "reward_total_composite_mean": 0.6973103880882263, "reward_total_composite_std": 0.2921626567840576} {"timestamp_utc": "2026-04-13T00:05:01Z", "mode": "train", "global_step": 753, "epoch": 0.07564038171772978, "loss": -0.0762, "grad_norm": 6.237549304962158, "learning_rate": 7.721212121212122e-06, "num_tokens": 1387549.0, "completions/mean_length": 101.25, "completions/min_length": 39.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 42.57143020629883, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.5288392305374146, "rewards/meter/std": 0.3813709318637848, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9938831925392151, "rewards/repeat_soft/std": 0.013261063024401665, "rewards/judge_quality/mean": 0.44749999046325684, "rewards/judge_quality/std": 0.23407875001430511, "rewards/total_composite/mean": 0.5437484979629517, "rewards/total_composite/std": 0.2942044734954834, "reward": 0.5437484979629517, "reward_std": 0.2942044734954834, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19106167554855347, "sampling/sampling_logp_difference/max": 1.229797601699829, "sampling/importance_sampling_ratio/min": 0.29235172271728516, "sampling/importance_sampling_ratio/mean": 1.0188486576080322, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3927685022354126, "clip_ratio/low_mean": 0.07332405634224415, "clip_ratio/low_min": 0.07332405634224415, "clip_ratio/high_mean": 0.08441830100491643, "clip_ratio/high_max": 0.08441830100491643, "clip_ratio/region_mean": 0.15774235734716058, "reward_total_mean": 0.5437484979629517, "reward_meter_mean": 0.5288392305374146, "reward_meter_std": 0.3813709318637848, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9938831925392151, "reward_repeat_soft_std": 0.013261063024401665, "reward_judge_quality_mean": 0.44749999046325684, "reward_judge_quality_std": 0.23407875001430511, "reward_total_composite_mean": 0.5437484979629517, "reward_total_composite_std": 0.2942044734954834} {"timestamp_utc": "2026-04-13T00:05:09Z", "mode": "train", "global_step": 754, "epoch": 0.07574083375188348, "loss": -0.0098, "grad_norm": 12.581169128417969, "learning_rate": 7.718181818181819e-06, "num_tokens": 1388958.0, "completions/mean_length": 31.125, "completions/min_length": 28.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.125, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.36368265748023987, "rewards/meter/std": 0.4041033983230591, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9991841316223145, "rewards/repeat_soft/std": 0.0018629790283739567, "rewards/judge_quality/mean": 0.5349999666213989, "rewards/judge_quality/std": 0.24663449823856354, "rewards/total_composite/mean": 0.5740756392478943, "rewards/total_composite/std": 0.15297575294971466, "reward": 0.5740756392478943, "reward_std": 0.15297576785087585, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11923474818468094, "sampling/sampling_logp_difference/max": 1.3591015338897705, "sampling/importance_sampling_ratio/min": 0.25689148902893066, "sampling/importance_sampling_ratio/mean": 1.011407732963562, "sampling/importance_sampling_ratio/max": 1.7436774969100952, "entropy": 0.7615501880645752, "clip_ratio/low_mean": 0.08328260481357574, "clip_ratio/low_min": 0.08328260481357574, "clip_ratio/high_mean": 0.031020894646644592, "clip_ratio/high_max": 0.031020894646644592, "clip_ratio/region_mean": 0.11430349946022034, "reward_total_mean": 0.5740756392478943, "reward_meter_mean": 0.36368265748023987, "reward_meter_std": 0.4041033983230591, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9991841316223145, "reward_repeat_soft_std": 0.0018629790283739567, "reward_judge_quality_mean": 0.5349999666213989, "reward_judge_quality_std": 0.24663449823856354, "reward_total_composite_mean": 0.5740756392478943, "reward_total_composite_std": 0.15297575294971466} {"timestamp_utc": "2026-04-13T00:05:16Z", "mode": "train", "global_step": 755, "epoch": 0.07584128578603717, "loss": 0.0306, "grad_norm": 18.393003463745117, "learning_rate": 7.715151515151516e-06, "num_tokens": 1390593.0, "completions/mean_length": 28.375, "completions/min_length": 24.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.375, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.7841629981994629, "rewards/meter/std": 0.2853557765483856, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9844369888305664, "rewards/repeat_soft/std": 0.01655511185526848, "rewards/judge_quality/mean": 0.8324999809265137, "rewards/judge_quality/std": 0.18077215552330017, "rewards/total_composite/mean": 0.851067066192627, "rewards/total_composite/std": 0.1209709495306015, "reward": 0.851067066192627, "reward_std": 0.1209709569811821, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1571994125843048, "sampling/sampling_logp_difference/max": 1.3267852067947388, "sampling/importance_sampling_ratio/min": 0.26532885432243347, "sampling/importance_sampling_ratio/mean": 1.0162928104400635, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.957863911986351, "clip_ratio/low_mean": 0.056197917088866234, "clip_ratio/low_min": 0.056197917088866234, "clip_ratio/high_mean": 0.09021786786615849, "clip_ratio/high_max": 0.09021786786615849, "clip_ratio/region_mean": 0.14641578495502472, "reward_total_mean": 0.851067066192627, "reward_meter_mean": 0.7841629981994629, "reward_meter_std": 0.2853557765483856, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9844369888305664, "reward_repeat_soft_std": 0.01655511185526848, "reward_judge_quality_mean": 0.8324999809265137, "reward_judge_quality_std": 0.18077215552330017, "reward_total_composite_mean": 0.851067066192627, "reward_total_composite_std": 0.1209709495306015} {"timestamp_utc": "2026-04-13T00:05:25Z", "mode": "train", "global_step": 756, "epoch": 0.07594173782019085, "loss": 0.0607, "grad_norm": 15.197017669677734, "learning_rate": 7.712121212121213e-06, "num_tokens": 1392312.0, "completions/mean_length": 47.875, "completions/min_length": 43.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.875, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.7795512080192566, "rewards/meter/std": 0.29094868898391724, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9942119121551514, "rewards/repeat_soft/std": 0.010068872943520546, "rewards/judge_quality/mean": 0.4437499940395355, "rewards/judge_quality/std": 0.2084594964981079, "rewards/total_composite/mean": 0.6780706644058228, "rewards/total_composite/std": 0.2830962836742401, "reward": 0.6780706644058228, "reward_std": 0.2830962836742401, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15402105450630188, "sampling/sampling_logp_difference/max": 1.385941982269287, "sampling/importance_sampling_ratio/min": 0.2500881254673004, "sampling/importance_sampling_ratio/mean": 1.0210673809051514, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.545622043311596, "clip_ratio/low_mean": 0.04816561937332153, "clip_ratio/low_min": 0.04816561937332153, "clip_ratio/high_mean": 0.10819007735699415, "clip_ratio/high_max": 0.10819007735699415, "clip_ratio/region_mean": 0.15635569673031569, "reward_total_mean": 0.6780706644058228, "reward_meter_mean": 0.7795512080192566, "reward_meter_std": 0.29094868898391724, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9942119121551514, "reward_repeat_soft_std": 0.010068872943520546, "reward_judge_quality_mean": 0.4437499940395355, "reward_judge_quality_std": 0.2084594964981079, "reward_total_composite_mean": 0.6780706644058228, "reward_total_composite_std": 0.2830962836742401} {"timestamp_utc": "2026-04-13T00:05:37Z", "mode": "train", "global_step": 757, "epoch": 0.07604218985434455, "loss": -0.0727, "grad_norm": 4.276841640472412, "learning_rate": 7.709090909090909e-06, "num_tokens": 1393829.0, "completions/mean_length": 91.625, "completions/min_length": 27.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 31.571430206298828, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.3206060528755188, "rewards/meter/std": 0.34102198481559753, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9757041931152344, "rewards/repeat_soft/std": 0.041977398097515106, "rewards/judge_quality/mean": 0.4437499940395355, "rewards/judge_quality/std": 0.23427321016788483, "rewards/total_composite/mean": 0.49175554513931274, "rewards/total_composite/std": 0.268668532371521, "reward": 0.49175554513931274, "reward_std": 0.268668532371521, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14951804280281067, "sampling/sampling_logp_difference/max": 1.1026136875152588, "sampling/importance_sampling_ratio/min": 0.332002192735672, "sampling/importance_sampling_ratio/mean": 1.0261478424072266, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.695130743086338, "clip_ratio/low_mean": 0.0763468760997057, "clip_ratio/low_min": 0.0763468760997057, "clip_ratio/high_mean": 0.07457172032445669, "clip_ratio/high_max": 0.07457172032445669, "clip_ratio/region_mean": 0.1509185964241624, "reward_total_mean": 0.49175554513931274, "reward_meter_mean": 0.3206060528755188, "reward_meter_std": 0.34102198481559753, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9757041931152344, "reward_repeat_soft_std": 0.041977398097515106, "reward_judge_quality_mean": 0.4437499940395355, "reward_judge_quality_std": 0.23427321016788483, "reward_total_composite_mean": 0.49175554513931274, "reward_total_composite_std": 0.268668532371521} {"timestamp_utc": "2026-04-13T00:05:44Z", "mode": "train", "global_step": 758, "epoch": 0.07614264188849824, "loss": 0.0004, "grad_norm": 18.039560317993164, "learning_rate": 7.706060606060606e-06, "num_tokens": 1395306.0, "completions/mean_length": 39.625, "completions/min_length": 32.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.625, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.5556830167770386, "rewards/meter/std": 0.44554561376571655, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.963543713092804, "rewards/repeat_soft/std": 0.04703747481107712, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.6246616840362549, "rewards/total_composite/std": 0.20113983750343323, "reward": 0.6246616840362549, "reward_std": 0.20113982260227203, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18429556488990784, "sampling/sampling_logp_difference/max": 1.464768409729004, "sampling/importance_sampling_ratio/min": 0.23113152384757996, "sampling/importance_sampling_ratio/mean": 1.0110770463943481, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2481596767902374, "clip_ratio/low_mean": 0.06354455463588238, "clip_ratio/low_min": 0.06354455463588238, "clip_ratio/high_mean": 0.08584279101341963, "clip_ratio/high_max": 0.08584279101341963, "clip_ratio/region_mean": 0.149387345649302, "reward_total_mean": 0.6246616840362549, "reward_meter_mean": 0.5556830167770386, "reward_meter_std": 0.44554561376571655, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.963543713092804, "reward_repeat_soft_std": 0.04703747481107712, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.6246616840362549, "reward_total_composite_std": 0.20113983750343323} {"timestamp_utc": "2026-04-13T00:05:51Z", "mode": "train", "global_step": 759, "epoch": 0.07624309392265194, "loss": -0.0178, "grad_norm": 10.743244171142578, "learning_rate": 7.703030303030304e-06, "num_tokens": 1397589.0, "completions/mean_length": 91.375, "completions/min_length": 80.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.375, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.4808642864227295, "rewards/meter/std": 0.4003073573112488, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9433087110519409, "rewards/repeat_soft/std": 0.06988590210676193, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.5792198181152344, "rewards/total_composite/std": 0.15662825107574463, "reward": 0.5792198181152344, "reward_std": 0.15662825107574463, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15516667068004608, "sampling/sampling_logp_difference/max": 1.3406884670257568, "sampling/importance_sampling_ratio/min": 0.2616654634475708, "sampling/importance_sampling_ratio/mean": 1.0073384046554565, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2308560237288475, "clip_ratio/low_mean": 0.0654004542157054, "clip_ratio/low_min": 0.0654004542157054, "clip_ratio/high_mean": 0.08080105856060982, "clip_ratio/high_max": 0.08080105856060982, "clip_ratio/region_mean": 0.1462015127763152, "reward_total_mean": 0.5792198181152344, "reward_meter_mean": 0.4808642864227295, "reward_meter_std": 0.4003073573112488, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9433087110519409, "reward_repeat_soft_std": 0.06988590210676193, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.5792198181152344, "reward_total_composite_std": 0.15662825107574463} {"timestamp_utc": "2026-04-13T00:06:04Z", "mode": "train", "global_step": 760, "epoch": 0.07634354595680562, "loss": -0.1533, "grad_norm": 4.479035377502441, "learning_rate": 7.7e-06, "num_tokens": 1399305.0, "completions/mean_length": 115.5, "completions/min_length": 32.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 58.857147216796875, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.6413788795471191, "rewards/meter/std": 0.48622095584869385, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.901417076587677, "rewards/repeat_soft/std": 0.20514753460884094, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.1345893144607544, "rewards/total_composite/mean": 0.5774584412574768, "rewards/total_composite/std": 0.29763463139533997, "reward": 0.5774584412574768, "reward_std": 0.29763466119766235, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1365724802017212, "sampling/sampling_logp_difference/max": 1.0287139415740967, "sampling/importance_sampling_ratio/min": 0.3574663996696472, "sampling/importance_sampling_ratio/mean": 1.0078917741775513, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9027073904871941, "clip_ratio/low_mean": 0.024218750186264515, "clip_ratio/low_min": 0.024218750186264515, "clip_ratio/high_mean": 0.07497742958366871, "clip_ratio/high_max": 0.07497742958366871, "clip_ratio/region_mean": 0.09919617976993322, "reward_total_mean": 0.5774584412574768, "reward_meter_mean": 0.6413788795471191, "reward_meter_std": 0.48622095584869385, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.901417076587677, "reward_repeat_soft_std": 0.20514753460884094, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.1345893144607544, "reward_total_composite_mean": 0.5774584412574768, "reward_total_composite_std": 0.29763463139533997} {"timestamp_utc": "2026-04-13T00:06:15Z", "mode": "train", "global_step": 761, "epoch": 0.07644399799095931, "loss": -0.2318, "grad_norm": 2.9747209548950195, "learning_rate": 7.696969696969696e-06, "num_tokens": 1401699.0, "completions/mean_length": 235.25, "completions/min_length": 96.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 143.0, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 206.0, "rewards/meter/mean": 0.9831862449645996, "rewards/meter/std": 0.017876267433166504, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.20773723721504211, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9111223816871643, "rewards/repeat_soft/std": 0.058787617832422256, "rewards/judge_quality/mean": 0.13750000298023224, "rewards/judge_quality/std": 0.06408699601888657, "rewards/total_composite/mean": 0.4476686120033264, "rewards/total_composite/std": 0.3711978495121002, "reward": 0.4476686120033264, "reward_std": 0.3711978495121002, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1550905853509903, "sampling/sampling_logp_difference/max": 3.362298011779785, "sampling/importance_sampling_ratio/min": 0.034655530005693436, "sampling/importance_sampling_ratio/mean": 1.035942792892456, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2672612965106964, "clip_ratio/low_mean": 0.02213541604578495, "clip_ratio/low_min": 0.02213541604578495, "clip_ratio/high_mean": 0.09257098380476236, "clip_ratio/high_max": 0.09257098380476236, "clip_ratio/region_mean": 0.11470639985054731, "reward_total_mean": 0.4476686120033264, "reward_meter_mean": 0.9831862449645996, "reward_meter_std": 0.017876267433166504, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.20773723721504211, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9111223816871643, "reward_repeat_soft_std": 0.058787617832422256, "reward_judge_quality_mean": 0.13750000298023224, "reward_judge_quality_std": 0.06408699601888657, "reward_total_composite_mean": 0.4476686120033264, "reward_total_composite_std": 0.3711978495121002} {"timestamp_utc": "2026-04-13T00:06:25Z", "mode": "train", "global_step": 762, "epoch": 0.07654445002511301, "loss": -0.0623, "grad_norm": 8.253721237182617, "learning_rate": 7.693939393939395e-06, "num_tokens": 1404249.0, "completions/mean_length": 108.75, "completions/min_length": 83.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.75, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.8868829607963562, "rewards/meter/std": 0.1804789900779724, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8041598796844482, "rewards/repeat_soft/std": 0.1364213526248932, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.7026382684707642, "rewards/total_composite/std": 0.06738372147083282, "reward": 0.7026382684707642, "reward_std": 0.06738372892141342, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1416543573141098, "sampling/sampling_logp_difference/max": 2.152553081512451, "sampling/importance_sampling_ratio/min": 0.11618714034557343, "sampling/importance_sampling_ratio/mean": 1.010478138923645, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0469124168157578, "clip_ratio/low_mean": 0.04688241798430681, "clip_ratio/low_min": 0.04688241798430681, "clip_ratio/high_mean": 0.06752835167571902, "clip_ratio/high_max": 0.06752835167571902, "clip_ratio/region_mean": 0.11441076966002584, "reward_total_mean": 0.7026382684707642, "reward_meter_mean": 0.8868829607963562, "reward_meter_std": 0.1804789900779724, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8041598796844482, "reward_repeat_soft_std": 0.1364213526248932, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.7026382684707642, "reward_total_composite_std": 0.06738372147083282} {"timestamp_utc": "2026-04-13T00:06:32Z", "mode": "train", "global_step": 763, "epoch": 0.0766449020592667, "loss": 0.0217, "grad_norm": 12.725303649902344, "learning_rate": 7.690909090909091e-06, "num_tokens": 1405896.0, "completions/mean_length": 45.875, "completions/min_length": 29.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.875, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.4666425585746765, "rewards/meter/std": 0.41779276728630066, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.11785111576318741, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9626809358596802, "rewards/repeat_soft/std": 0.05097004771232605, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.5595072507858276, "rewards/total_composite/std": 0.2127610594034195, "reward": 0.5595072507858276, "reward_std": 0.2127610594034195, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16727417707443237, "sampling/sampling_logp_difference/max": 2.5699477195739746, "sampling/importance_sampling_ratio/min": 0.076539546251297, "sampling/importance_sampling_ratio/mean": 1.0007987022399902, "sampling/importance_sampling_ratio/max": 1.9376615285873413, "entropy": 1.0572734475135803, "clip_ratio/low_mean": 0.07618686370551586, "clip_ratio/low_min": 0.07618686370551586, "clip_ratio/high_mean": 0.0626103077083826, "clip_ratio/high_max": 0.0626103077083826, "clip_ratio/region_mean": 0.13879717141389847, "reward_total_mean": 0.5595072507858276, "reward_meter_mean": 0.4666425585746765, "reward_meter_std": 0.41779276728630066, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.11785111576318741, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9626809358596802, "reward_repeat_soft_std": 0.05097004771232605, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.5595072507858276, "reward_total_composite_std": 0.2127610594034195} {"timestamp_utc": "2026-04-13T00:06:44Z", "mode": "train", "global_step": 764, "epoch": 0.0767453540934204, "loss": 0.1852, "grad_norm": 5.922009468078613, "learning_rate": 7.687878787878788e-06, "num_tokens": 1407615.0, "completions/mean_length": 133.875, "completions/min_length": 38.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 79.85714721679688, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 220.0, "rewards/meter/mean": 0.8688684105873108, "rewards/meter/std": 0.34392765164375305, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9013789892196655, "rewards/repeat_soft/std": 0.07774077355861664, "rewards/judge_quality/mean": 0.2587500214576721, "rewards/judge_quality/std": 0.13953366875648499, "rewards/total_composite/mean": 0.5855386853218079, "rewards/total_composite/std": 0.36243122816085815, "reward": 0.5855386853218079, "reward_std": 0.36243122816085815, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16742563247680664, "sampling/sampling_logp_difference/max": 1.7384519577026367, "sampling/importance_sampling_ratio/min": 0.17579232156276703, "sampling/importance_sampling_ratio/mean": 1.023047924041748, "sampling/importance_sampling_ratio/max": 1.9596796035766602, "entropy": 1.1842331662774086, "clip_ratio/low_mean": 0.010795454494655132, "clip_ratio/low_min": 0.010795454494655132, "clip_ratio/high_mean": 0.10508371517062187, "clip_ratio/high_max": 0.10508371517062187, "clip_ratio/region_mean": 0.115879169665277, "reward_total_mean": 0.5855386853218079, "reward_meter_mean": 0.8688684105873108, "reward_meter_std": 0.34392765164375305, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9013789892196655, "reward_repeat_soft_std": 0.07774077355861664, "reward_judge_quality_mean": 0.2587500214576721, "reward_judge_quality_std": 0.13953366875648499, "reward_total_composite_mean": 0.5855386853218079, "reward_total_composite_std": 0.36243122816085815} {"timestamp_utc": "2026-04-13T00:06:51Z", "mode": "train", "global_step": 765, "epoch": 0.07684580612757408, "loss": 0.0227, "grad_norm": 17.86490249633789, "learning_rate": 7.684848484848485e-06, "num_tokens": 1409034.0, "completions/mean_length": 28.375, "completions/min_length": 17.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.375, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.8142471313476562, "rewards/meter/std": 0.3392540216445923, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9603556990623474, "rewards/repeat_soft/std": 0.011151899583637714, "rewards/judge_quality/mean": 0.3687500059604645, "rewards/judge_quality/std": 0.10802611708641052, "rewards/total_composite/mean": 0.723071813583374, "rewards/total_composite/std": 0.15524591505527496, "reward": 0.723071813583374, "reward_std": 0.15524592995643616, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20683348178863525, "sampling/sampling_logp_difference/max": 1.8099112510681152, "sampling/importance_sampling_ratio/min": 0.1636686623096466, "sampling/importance_sampling_ratio/mean": 1.0109316110610962, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.092111200094223, "clip_ratio/low_mean": 0.051733193919062614, "clip_ratio/low_min": 0.051733193919062614, "clip_ratio/high_mean": 0.17655950412154198, "clip_ratio/high_max": 0.17655950412154198, "clip_ratio/region_mean": 0.2282926980406046, "reward_total_mean": 0.723071813583374, "reward_meter_mean": 0.8142471313476562, "reward_meter_std": 0.3392540216445923, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9603556990623474, "reward_repeat_soft_std": 0.011151899583637714, "reward_judge_quality_mean": 0.3687500059604645, "reward_judge_quality_std": 0.10802611708641052, "reward_total_composite_mean": 0.723071813583374, "reward_total_composite_std": 0.15524591505527496} {"timestamp_utc": "2026-04-13T00:06:58Z", "mode": "train", "global_step": 766, "epoch": 0.07694625816172777, "loss": 0.0162, "grad_norm": 15.099808692932129, "learning_rate": 7.681818181818183e-06, "num_tokens": 1410838.0, "completions/mean_length": 47.5, "completions/min_length": 40.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.5, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.9640952348709106, "rewards/meter/std": 0.0585465282201767, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9403992891311646, "rewards/repeat_soft/std": 0.0870438739657402, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.19255799055099487, "rewards/total_composite/mean": 0.6308404803276062, "rewards/total_composite/std": 0.39317384362220764, "reward": 0.6308404803276062, "reward_std": 0.39317384362220764, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13000726699829102, "sampling/sampling_logp_difference/max": 1.336362361907959, "sampling/importance_sampling_ratio/min": 0.2627999186515808, "sampling/importance_sampling_ratio/mean": 1.0089401006698608, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0111509710550308, "clip_ratio/low_mean": 0.01902173925191164, "clip_ratio/low_min": 0.01902173925191164, "clip_ratio/high_mean": 0.1096693528816104, "clip_ratio/high_max": 0.1096693528816104, "clip_ratio/region_mean": 0.12869109213352203, "reward_total_mean": 0.6308404803276062, "reward_meter_mean": 0.9640952348709106, "reward_meter_std": 0.0585465282201767, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9403992891311646, "reward_repeat_soft_std": 0.0870438739657402, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.19255799055099487, "reward_total_composite_mean": 0.6308404803276062, "reward_total_composite_std": 0.39317384362220764} {"timestamp_utc": "2026-04-13T00:07:09Z", "mode": "train", "global_step": 767, "epoch": 0.07704671019588147, "loss": 0.0468, "grad_norm": 15.304726600646973, "learning_rate": 7.678787878787878e-06, "num_tokens": 1412711.0, "completions/mean_length": 65.125, "completions/min_length": 54.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.125, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.5315076112747192, "rewards/meter/std": 0.21202845871448517, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9519174695014954, "rewards/repeat_soft/std": 0.05758032575249672, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.6009951829910278, "rewards/total_composite/std": 0.10237671434879303, "reward": 0.6009951829910278, "reward_std": 0.10237671434879303, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1736992448568344, "sampling/sampling_logp_difference/max": 1.923752784729004, "sampling/importance_sampling_ratio/min": 0.14605781435966492, "sampling/importance_sampling_ratio/mean": 1.0042340755462646, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.097369559109211, "clip_ratio/low_mean": 0.09881812892854214, "clip_ratio/low_min": 0.09881812892854214, "clip_ratio/high_mean": 0.06997177749872208, "clip_ratio/high_max": 0.06997177749872208, "clip_ratio/region_mean": 0.1687899064272642, "reward_total_mean": 0.6009951829910278, "reward_meter_mean": 0.5315076112747192, "reward_meter_std": 0.21202845871448517, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9519174695014954, "reward_repeat_soft_std": 0.05758032575249672, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.6009951829910278, "reward_total_composite_std": 0.10237671434879303} {"timestamp_utc": "2026-04-13T00:07:17Z", "mode": "train", "global_step": 768, "epoch": 0.07714716223003516, "loss": 0.027, "grad_norm": 22.731725692749023, "learning_rate": 7.675757575757577e-06, "num_tokens": 1414307.0, "completions/mean_length": 30.5, "completions/min_length": 26.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.5, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.6463751792907715, "rewards/meter/std": 0.41417959332466125, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9927520751953125, "rewards/repeat_soft/std": 0.0144913075491786, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.741144061088562, "rewards/total_composite/std": 0.25487855076789856, "reward": 0.741144061088562, "reward_std": 0.25487855076789856, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1705755740404129, "sampling/sampling_logp_difference/max": 2.7310943603515625, "sampling/importance_sampling_ratio/min": 0.06514795869588852, "sampling/importance_sampling_ratio/mean": 1.0047637224197388, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7716916725039482, "clip_ratio/low_mean": 0.05162151250988245, "clip_ratio/low_min": 0.05162151250988245, "clip_ratio/high_mean": 0.07495181635022163, "clip_ratio/high_max": 0.07495181635022163, "clip_ratio/region_mean": 0.12657332886010408, "reward_total_mean": 0.741144061088562, "reward_meter_mean": 0.6463751792907715, "reward_meter_std": 0.41417959332466125, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9927520751953125, "reward_repeat_soft_std": 0.0144913075491786, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.741144061088562, "reward_total_composite_std": 0.25487855076789856} {"timestamp_utc": "2026-04-13T00:07:31Z", "mode": "train", "global_step": 769, "epoch": 0.07724761426418884, "loss": -0.0574, "grad_norm": 2.976480484008789, "learning_rate": 7.672727272727273e-06, "num_tokens": 1415654.0, "completions/mean_length": 80.375, "completions/min_length": 14.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 18.71428680419922, "completions/min_terminated_length": 14.0, "completions/max_terminated_length": 22.0, "rewards/meter/mean": 0.7267776727676392, "rewards/meter/std": 0.371418297290802, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9250398874282837, "rewards/repeat_soft/std": 0.03461219370365143, "rewards/judge_quality/mean": 0.3475000262260437, "rewards/judge_quality/std": 0.27243608236312866, "rewards/total_composite/mean": 0.6371285915374756, "rewards/total_composite/std": 0.29745936393737793, "reward": 0.6371285915374756, "reward_std": 0.29745936393737793, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1634591966867447, "sampling/sampling_logp_difference/max": 3.248734474182129, "sampling/importance_sampling_ratio/min": 0.03882330656051636, "sampling/importance_sampling_ratio/mean": 1.003374457359314, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5431215837597847, "clip_ratio/low_mean": 0.04707792401313782, "clip_ratio/low_min": 0.04707792401313782, "clip_ratio/high_mean": 0.09614108875393867, "clip_ratio/high_max": 0.09614108875393867, "clip_ratio/region_mean": 0.1432190127670765, "reward_total_mean": 0.6371285915374756, "reward_meter_mean": 0.7267776727676392, "reward_meter_std": 0.371418297290802, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9250398874282837, "reward_repeat_soft_std": 0.03461219370365143, "reward_judge_quality_mean": 0.3475000262260437, "reward_judge_quality_std": 0.27243608236312866, "reward_total_composite_mean": 0.6371285915374756, "reward_total_composite_std": 0.29745936393737793} {"timestamp_utc": "2026-04-13T00:07:43Z", "mode": "train", "global_step": 770, "epoch": 0.07734806629834254, "loss": 0.1259, "grad_norm": 18.812938690185547, "learning_rate": 7.66969696969697e-06, "num_tokens": 1417402.0, "completions/mean_length": 57.5, "completions/min_length": 48.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.5, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.8615058660507202, "rewards/meter/std": 0.2873074412345886, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9358267188072205, "rewards/repeat_soft/std": 0.07799286395311356, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.7321352958679199, "rewards/total_composite/std": 0.09583711624145508, "reward": 0.7321352958679199, "reward_std": 0.09583713114261627, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15432612597942352, "sampling/sampling_logp_difference/max": 0.942084789276123, "sampling/importance_sampling_ratio/min": 0.3898142874240875, "sampling/importance_sampling_ratio/mean": 1.0348013639450073, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2777834832668304, "clip_ratio/low_mean": 0.0313967140391469, "clip_ratio/low_min": 0.0313967140391469, "clip_ratio/high_mean": 0.12649431824684143, "clip_ratio/high_max": 0.12649431824684143, "clip_ratio/region_mean": 0.15789103228598833, "reward_total_mean": 0.7321352958679199, "reward_meter_mean": 0.8615058660507202, "reward_meter_std": 0.2873074412345886, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9358267188072205, "reward_repeat_soft_std": 0.07799286395311356, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.7321352958679199, "reward_total_composite_std": 0.09583711624145508} {"timestamp_utc": "2026-04-13T00:07:51Z", "mode": "train", "global_step": 771, "epoch": 0.07744851833249623, "loss": 0.0825, "grad_norm": 17.610116958618164, "learning_rate": 7.666666666666667e-06, "num_tokens": 1418917.0, "completions/mean_length": 39.375, "completions/min_length": 31.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.375, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.786068320274353, "rewards/meter/std": 0.2902088761329651, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.998836874961853, "rewards/repeat_soft/std": 0.001946197240613401, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7329893708229065, "rewards/total_composite/std": 0.12946569919586182, "reward": 0.7329893708229065, "reward_std": 0.12946568429470062, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16631412506103516, "sampling/sampling_logp_difference/max": 3.318559169769287, "sampling/importance_sampling_ratio/min": 0.03620496019721031, "sampling/importance_sampling_ratio/mean": 1.0199309587478638, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0413255542516708, "clip_ratio/low_mean": 0.04166235402226448, "clip_ratio/low_min": 0.04166235402226448, "clip_ratio/high_mean": 0.11483925674110651, "clip_ratio/high_max": 0.11483925674110651, "clip_ratio/region_mean": 0.156501610763371, "reward_total_mean": 0.7329893708229065, "reward_meter_mean": 0.786068320274353, "reward_meter_std": 0.2902088761329651, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.998836874961853, "reward_repeat_soft_std": 0.001946197240613401, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7329893708229065, "reward_total_composite_std": 0.12946569919586182} {"timestamp_utc": "2026-04-13T00:08:03Z", "mode": "train", "global_step": 772, "epoch": 0.07754897036664993, "loss": -0.0651, "grad_norm": 1.459313988685608, "learning_rate": 7.663636363636364e-06, "num_tokens": 1420244.0, "completions/mean_length": 80.875, "completions/min_length": 15.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 19.285715103149414, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.8096297979354858, "rewards/meter/std": 0.3537612855434418, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9031335115432739, "rewards/repeat_soft/std": 0.14878186583518982, "rewards/judge_quality/mean": 0.3387499749660492, "rewards/judge_quality/std": 0.14327171444892883, "rewards/total_composite/mean": 0.6727868318557739, "rewards/total_composite/std": 0.2812933623790741, "reward": 0.6727868318557739, "reward_std": 0.2812933623790741, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12346980720758438, "sampling/sampling_logp_difference/max": 1.9033327102661133, "sampling/importance_sampling_ratio/min": 0.14907099306583405, "sampling/importance_sampling_ratio/mean": 1.034257411956787, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8461377248167992, "clip_ratio/low_mean": 0.029411764815449715, "clip_ratio/low_min": 0.029411764815449715, "clip_ratio/high_mean": 0.07412994932383299, "clip_ratio/high_max": 0.07412994932383299, "clip_ratio/region_mean": 0.1035417141392827, "reward_total_mean": 0.6727868318557739, "reward_meter_mean": 0.8096297979354858, "reward_meter_std": 0.3537612855434418, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9031335115432739, "reward_repeat_soft_std": 0.14878186583518982, "reward_judge_quality_mean": 0.3387499749660492, "reward_judge_quality_std": 0.14327171444892883, "reward_total_composite_mean": 0.6727868318557739, "reward_total_composite_std": 0.2812933623790741} {"timestamp_utc": "2026-04-13T00:08:12Z", "mode": "train", "global_step": 773, "epoch": 0.07764942240080362, "loss": 0.1076, "grad_norm": 11.08802604675293, "learning_rate": 7.660606060606062e-06, "num_tokens": 1422059.0, "completions/mean_length": 55.875, "completions/min_length": 34.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.875, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9004024267196655, "rewards/meter/std": 0.20440968871116638, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8224330544471741, "rewards/repeat_soft/std": 0.2238134890794754, "rewards/judge_quality/mean": 0.2887499928474426, "rewards/judge_quality/std": 0.11630470305681229, "rewards/total_composite/mean": 0.7240493893623352, "rewards/total_composite/std": 0.07652736455202103, "reward": 0.7240493893623352, "reward_std": 0.07652736455202103, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14684687554836273, "sampling/sampling_logp_difference/max": 2.7947521209716797, "sampling/importance_sampling_ratio/min": 0.06113002449274063, "sampling/importance_sampling_ratio/mean": 1.0110530853271484, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0014852806925774, "clip_ratio/low_mean": 0.039230383932590485, "clip_ratio/low_min": 0.039230383932590485, "clip_ratio/high_mean": 0.10596537310630083, "clip_ratio/high_max": 0.10596537310630083, "clip_ratio/region_mean": 0.14519575703889132, "reward_total_mean": 0.7240493893623352, "reward_meter_mean": 0.9004024267196655, "reward_meter_std": 0.20440968871116638, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8224330544471741, "reward_repeat_soft_std": 0.2238134890794754, "reward_judge_quality_mean": 0.2887499928474426, "reward_judge_quality_std": 0.11630470305681229, "reward_total_composite_mean": 0.7240493893623352, "reward_total_composite_std": 0.07652736455202103} {"timestamp_utc": "2026-04-13T00:08:21Z", "mode": "train", "global_step": 774, "epoch": 0.0777498744349573, "loss": 0.0116, "grad_norm": 11.876138687133789, "learning_rate": 7.657575757575757e-06, "num_tokens": 1423731.0, "completions/mean_length": 45.0, "completions/min_length": 40.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.0, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.9900963306427002, "rewards/meter/std": 0.004402524325996637, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9351577758789062, "rewards/repeat_soft/std": 0.05734028294682503, "rewards/judge_quality/mean": 0.45249998569488525, "rewards/judge_quality/std": 0.21224986016750336, "rewards/total_composite/mean": 0.824809193611145, "rewards/total_composite/std": 0.0662183165550232, "reward": 0.824809193611145, "reward_std": 0.06621832400560379, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12863123416900635, "sampling/sampling_logp_difference/max": 1.5798730850219727, "sampling/importance_sampling_ratio/min": 0.20600123703479767, "sampling/importance_sampling_ratio/mean": 1.0337836742401123, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8615250736474991, "clip_ratio/low_mean": 0.08479641168378294, "clip_ratio/low_min": 0.08479641168378294, "clip_ratio/high_mean": 0.02361111156642437, "clip_ratio/high_max": 0.02361111156642437, "clip_ratio/region_mean": 0.1084075232502073, "reward_total_mean": 0.824809193611145, "reward_meter_mean": 0.9900963306427002, "reward_meter_std": 0.004402524325996637, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9351577758789062, "reward_repeat_soft_std": 0.05734028294682503, "reward_judge_quality_mean": 0.45249998569488525, "reward_judge_quality_std": 0.21224986016750336, "reward_total_composite_mean": 0.824809193611145, "reward_total_composite_std": 0.0662183165550232} {"timestamp_utc": "2026-04-13T00:08:30Z", "mode": "train", "global_step": 775, "epoch": 0.077850326469111, "loss": 0.0808, "grad_norm": 13.628656387329102, "learning_rate": 7.654545454545456e-06, "num_tokens": 1425352.0, "completions/mean_length": 36.625, "completions/min_length": 31.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.625, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.9821764826774597, "rewards/meter/std": 0.005196631886065006, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.978819727897644, "rewards/repeat_soft/std": 0.014733008109033108, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.8357363939285278, "rewards/total_composite/std": 0.054432667791843414, "reward": 0.8357363939285278, "reward_std": 0.05443267151713371, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1691982001066208, "sampling/sampling_logp_difference/max": 1.8255224227905273, "sampling/importance_sampling_ratio/min": 0.22910206019878387, "sampling/importance_sampling_ratio/mean": 1.0153117179870605, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.109547197818756, "clip_ratio/low_mean": 0.12806707248091698, "clip_ratio/low_min": 0.12806707248091698, "clip_ratio/high_mean": 0.024193547666072845, "clip_ratio/high_max": 0.024193547666072845, "clip_ratio/region_mean": 0.15226062014698982, "reward_total_mean": 0.8357363939285278, "reward_meter_mean": 0.9821764826774597, "reward_meter_std": 0.005196631886065006, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.978819727897644, "reward_repeat_soft_std": 0.014733008109033108, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.8357363939285278, "reward_total_composite_std": 0.054432667791843414} {"timestamp_utc": "2026-04-13T00:08:38Z", "mode": "train", "global_step": 776, "epoch": 0.07795077850326469, "loss": 0.0538, "grad_norm": 10.586370468139648, "learning_rate": 7.651515151515152e-06, "num_tokens": 1426999.0, "completions/mean_length": 61.875, "completions/min_length": 56.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.875, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9827167391777039, "rewards/meter/std": 0.02089310623705387, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8891125917434692, "rewards/repeat_soft/std": 0.13650958240032196, "rewards/judge_quality/mean": 0.2574999928474426, "rewards/judge_quality/std": 0.14078859984874725, "rewards/total_composite/mean": 0.7583838105201721, "rewards/total_composite/std": 0.05608035624027252, "reward": 0.7583838105201721, "reward_std": 0.05608034133911133, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12488245218992233, "sampling/sampling_logp_difference/max": 1.4151325225830078, "sampling/importance_sampling_ratio/min": 0.24289342761039734, "sampling/importance_sampling_ratio/mean": 1.0346640348434448, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0396185666322708, "clip_ratio/low_mean": 0.055309977382421494, "clip_ratio/low_min": 0.055309977382421494, "clip_ratio/high_mean": 0.046003447379916906, "clip_ratio/high_max": 0.046003447379916906, "clip_ratio/region_mean": 0.1013134247623384, "reward_total_mean": 0.7583838105201721, "reward_meter_mean": 0.9827167391777039, "reward_meter_std": 0.02089310623705387, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8891125917434692, "reward_repeat_soft_std": 0.13650958240032196, "reward_judge_quality_mean": 0.2574999928474426, "reward_judge_quality_std": 0.14078859984874725, "reward_total_composite_mean": 0.7583838105201721, "reward_total_composite_std": 0.05608035624027252} {"timestamp_utc": "2026-04-13T00:08:48Z", "mode": "train", "global_step": 777, "epoch": 0.07805123053741839, "loss": 0.0764, "grad_norm": 17.20375633239746, "learning_rate": 7.648484848484849e-06, "num_tokens": 1428483.0, "completions/mean_length": 22.5, "completions/min_length": 19.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.5, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.9407755732536316, "rewards/meter/std": 0.08982627838850021, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9621384143829346, "rewards/repeat_soft/std": 0.0010226722806692123, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.8154378533363342, "rewards/total_composite/std": 0.07311612367630005, "reward": 0.8154378533363342, "reward_std": 0.07311611622571945, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17128968238830566, "sampling/sampling_logp_difference/max": 2.104130268096924, "sampling/importance_sampling_ratio/min": 0.3122539222240448, "sampling/importance_sampling_ratio/mean": 1.0236026048660278, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0877759456634521, "clip_ratio/low_mean": 0.05785573087632656, "clip_ratio/low_min": 0.05785573087632656, "clip_ratio/high_mean": 0.1285785511136055, "clip_ratio/high_max": 0.1285785511136055, "clip_ratio/region_mean": 0.18643428198993206, "reward_total_mean": 0.8154378533363342, "reward_meter_mean": 0.9407755732536316, "reward_meter_std": 0.08982627838850021, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9621384143829346, "reward_repeat_soft_std": 0.0010226722806692123, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.8154378533363342, "reward_total_composite_std": 0.07311612367630005} {"timestamp_utc": "2026-04-13T00:08:56Z", "mode": "train", "global_step": 778, "epoch": 0.07815168257157207, "loss": 0.0614, "grad_norm": 11.537582397460938, "learning_rate": 7.645454545454546e-06, "num_tokens": 1430420.0, "completions/mean_length": 62.125, "completions/min_length": 41.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.125, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.7539616823196411, "rewards/meter/std": 0.2771271765232086, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.17817415297031403, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9694095849990845, "rewards/repeat_soft/std": 0.03241977468132973, "rewards/judge_quality/mean": 0.4475000202655792, "rewards/judge_quality/std": 0.1381252110004425, "rewards/total_composite/mean": 0.6954736709594727, "rewards/total_composite/std": 0.12091825157403946, "reward": 0.6954736709594727, "reward_std": 0.12091824412345886, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17406116425991058, "sampling/sampling_logp_difference/max": 1.7056169509887695, "sampling/importance_sampling_ratio/min": 0.18166027963161469, "sampling/importance_sampling_ratio/mean": 1.024019479751587, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5232604891061783, "clip_ratio/low_mean": 0.05000685527920723, "clip_ratio/low_min": 0.05000685527920723, "clip_ratio/high_mean": 0.09911742061376572, "clip_ratio/high_max": 0.09911742061376572, "clip_ratio/region_mean": 0.14912427589297295, "reward_total_mean": 0.6954736709594727, "reward_meter_mean": 0.7539616823196411, "reward_meter_std": 0.2771271765232086, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.17817415297031403, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9694095849990845, "reward_repeat_soft_std": 0.03241977468132973, "reward_judge_quality_mean": 0.4475000202655792, "reward_judge_quality_std": 0.1381252110004425, "reward_total_composite_mean": 0.6954736709594727, "reward_total_composite_std": 0.12091825157403946} {"timestamp_utc": "2026-04-13T00:09:03Z", "mode": "train", "global_step": 779, "epoch": 0.07825213460572576, "loss": 0.0622, "grad_norm": 18.70929527282715, "learning_rate": 7.642424242424244e-06, "num_tokens": 1432086.0, "completions/mean_length": 40.25, "completions/min_length": 36.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.25, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.8219190835952759, "rewards/meter/std": 0.2903538942337036, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9579335451126099, "rewards/repeat_soft/std": 0.03299632668495178, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7450319528579712, "rewards/total_composite/std": 0.13086426258087158, "reward": 0.7450319528579712, "reward_std": 0.13086427748203278, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17843644320964813, "sampling/sampling_logp_difference/max": 1.70562744140625, "sampling/importance_sampling_ratio/min": 0.18165837228298187, "sampling/importance_sampling_ratio/mean": 1.0161561965942383, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1758508533239365, "clip_ratio/low_mean": 0.01718073640950024, "clip_ratio/low_min": 0.01718073640950024, "clip_ratio/high_mean": 0.12444119714200497, "clip_ratio/high_max": 0.12444119714200497, "clip_ratio/region_mean": 0.1416219335515052, "reward_total_mean": 0.7450319528579712, "reward_meter_mean": 0.8219190835952759, "reward_meter_std": 0.2903538942337036, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9579335451126099, "reward_repeat_soft_std": 0.03299632668495178, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7450319528579712, "reward_total_composite_std": 0.13086426258087158} {"timestamp_utc": "2026-04-13T00:09:18Z", "mode": "train", "global_step": 780, "epoch": 0.07835258663987946, "loss": -0.1044, "grad_norm": 3.046682119369507, "learning_rate": 7.639393939393939e-06, "num_tokens": 1433833.0, "completions/mean_length": 108.375, "completions/min_length": 43.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 50.71428680419922, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.7343107461929321, "rewards/meter/std": 0.32804977893829346, "rewards/count_adherence/mean": 0.71875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9330458641052246, "rewards/repeat_soft/std": 0.058578480035066605, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.28965190052986145, "rewards/total_composite/mean": 0.6551505327224731, "rewards/total_composite/std": 0.29258865118026733, "reward": 0.6551505327224731, "reward_std": 0.29258862137794495, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15078242123126984, "sampling/sampling_logp_difference/max": 1.4056119918823242, "sampling/importance_sampling_ratio/min": 0.24521693587303162, "sampling/importance_sampling_ratio/mean": 1.039799690246582, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1120755895972252, "clip_ratio/low_mean": 0.028767338953912258, "clip_ratio/low_min": 0.028767338953912258, "clip_ratio/high_mean": 0.08599704131484032, "clip_ratio/high_max": 0.08599704131484032, "clip_ratio/region_mean": 0.11476438026875257, "reward_total_mean": 0.6551505327224731, "reward_meter_mean": 0.7343107461929321, "reward_meter_std": 0.32804977893829346, "reward_count_adherence_mean": 0.71875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9330458641052246, "reward_repeat_soft_std": 0.058578480035066605, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.28965190052986145, "reward_total_composite_mean": 0.6551505327224731, "reward_total_composite_std": 0.29258865118026733} {"timestamp_utc": "2026-04-13T00:09:27Z", "mode": "train", "global_step": 781, "epoch": 0.07845303867403315, "loss": 0.0384, "grad_norm": 15.706035614013672, "learning_rate": 7.636363636363638e-06, "num_tokens": 1435429.0, "completions/mean_length": 49.5, "completions/min_length": 38.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.5, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.781302809715271, "rewards/meter/std": 0.32292479276657104, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9794652462005615, "rewards/repeat_soft/std": 0.026740342378616333, "rewards/judge_quality/mean": 0.4675000011920929, "rewards/judge_quality/std": 0.21022097766399384, "rewards/total_composite/mean": 0.7397828102111816, "rewards/total_composite/std": 0.1698509305715561, "reward": 0.7397828102111816, "reward_std": 0.1698509305715561, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1735345721244812, "sampling/sampling_logp_difference/max": 1.3637161254882812, "sampling/importance_sampling_ratio/min": 0.2557087540626526, "sampling/importance_sampling_ratio/mean": 1.023484230041504, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3505277633666992, "clip_ratio/low_mean": 0.047327364794909954, "clip_ratio/low_min": 0.047327364794909954, "clip_ratio/high_mean": 0.09129845909774303, "clip_ratio/high_max": 0.09129845909774303, "clip_ratio/region_mean": 0.138625823892653, "reward_total_mean": 0.7397828102111816, "reward_meter_mean": 0.781302809715271, "reward_meter_std": 0.32292479276657104, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9794652462005615, "reward_repeat_soft_std": 0.026740342378616333, "reward_judge_quality_mean": 0.4675000011920929, "reward_judge_quality_std": 0.21022097766399384, "reward_total_composite_mean": 0.7397828102111816, "reward_total_composite_std": 0.1698509305715561} {"timestamp_utc": "2026-04-13T00:09:35Z", "mode": "train", "global_step": 782, "epoch": 0.07855349070818685, "loss": 0.1781, "grad_norm": 25.80482292175293, "learning_rate": 7.633333333333334e-06, "num_tokens": 1436785.0, "completions/mean_length": 19.5, "completions/min_length": 17.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.5, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.7282249927520752, "rewards/meter/std": 0.36287564039230347, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.5637500286102295, "rewards/judge_quality/std": 0.22012579441070557, "rewards/total_composite/mean": 0.7430762648582458, "rewards/total_composite/std": 0.19253799319267273, "reward": 0.7430762648582458, "reward_std": 0.19253802299499512, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21488052606582642, "sampling/sampling_logp_difference/max": 1.783494472503662, "sampling/importance_sampling_ratio/min": 0.1680498719215393, "sampling/importance_sampling_ratio/mean": 1.0449156761169434, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4354719147086143, "clip_ratio/low_mean": 0.052678571082651615, "clip_ratio/low_min": 0.052678571082651615, "clip_ratio/high_mean": 0.15306956321001053, "clip_ratio/high_max": 0.15306956321001053, "clip_ratio/region_mean": 0.20574813429266214, "reward_total_mean": 0.7430762648582458, "reward_meter_mean": 0.7282249927520752, "reward_meter_std": 0.36287564039230347, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.5637500286102295, "reward_judge_quality_std": 0.22012579441070557, "reward_total_composite_mean": 0.7430762648582458, "reward_total_composite_std": 0.19253799319267273} {"timestamp_utc": "2026-04-13T00:09:50Z", "mode": "train", "global_step": 783, "epoch": 0.07865394274234053, "loss": -0.1068, "grad_norm": 3.2029290199279785, "learning_rate": 7.630303030303031e-06, "num_tokens": 1438426.0, "completions/mean_length": 102.125, "completions/min_length": 40.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 43.57143020629883, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.8788701295852661, "rewards/meter/std": 0.26476040482521057, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9382221698760986, "rewards/repeat_soft/std": 0.08937608450651169, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.13845446705818176, "rewards/total_composite/mean": 0.6628350019454956, "rewards/total_composite/std": 0.29479074478149414, "reward": 0.6628350019454956, "reward_std": 0.29479071497917175, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16152837872505188, "sampling/sampling_logp_difference/max": 2.377847194671631, "sampling/importance_sampling_ratio/min": 0.09275003522634506, "sampling/importance_sampling_ratio/mean": 1.030977487564087, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.266288161277771, "clip_ratio/low_mean": 0.015306122601032257, "clip_ratio/low_min": 0.015306122601032257, "clip_ratio/high_mean": 0.11705856211483479, "clip_ratio/high_max": 0.11705856211483479, "clip_ratio/region_mean": 0.13236468471586704, "reward_total_mean": 0.6628350019454956, "reward_meter_mean": 0.8788701295852661, "reward_meter_std": 0.26476040482521057, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9382221698760986, "reward_repeat_soft_std": 0.08937608450651169, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.13845446705818176, "reward_total_composite_mean": 0.6628350019454956, "reward_total_composite_std": 0.29479074478149414} {"timestamp_utc": "2026-04-13T00:10:05Z", "mode": "train", "global_step": 784, "epoch": 0.07875439477649422, "loss": -0.0756, "grad_norm": 3.639958620071411, "learning_rate": 7.627272727272727e-06, "num_tokens": 1440206.0, "completions/mean_length": 166.5, "completions/min_length": 39.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 51.333335876464844, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.8622908592224121, "rewards/meter/std": 0.3484167754650116, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.2519763112068176, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9379162788391113, "rewards/repeat_soft/std": 0.056217215955257416, "rewards/judge_quality/mean": 0.24249999225139618, "rewards/judge_quality/std": 0.14007650315761566, "rewards/total_composite/mean": 0.49028337001800537, "rewards/total_composite/std": 0.36523932218551636, "reward": 0.49028337001800537, "reward_std": 0.36523929238319397, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19368159770965576, "sampling/sampling_logp_difference/max": 1.9181780815124512, "sampling/importance_sampling_ratio/min": 0.1468743085861206, "sampling/importance_sampling_ratio/mean": 1.0267062187194824, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0953802913427353, "clip_ratio/low_mean": 0.016826923936605453, "clip_ratio/low_min": 0.016826923936605453, "clip_ratio/high_mean": 0.12682975642383099, "clip_ratio/high_max": 0.12682975642383099, "clip_ratio/region_mean": 0.14365668036043644, "reward_total_mean": 0.49028337001800537, "reward_meter_mean": 0.8622908592224121, "reward_meter_std": 0.3484167754650116, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.2519763112068176, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9379162788391113, "reward_repeat_soft_std": 0.056217215955257416, "reward_judge_quality_mean": 0.24249999225139618, "reward_judge_quality_std": 0.14007650315761566, "reward_total_composite_mean": 0.49028337001800537, "reward_total_composite_std": 0.36523932218551636} {"timestamp_utc": "2026-04-13T00:10:13Z", "mode": "train", "global_step": 785, "epoch": 0.07885484681064792, "loss": -0.0503, "grad_norm": 13.491278648376465, "learning_rate": 7.6242424242424254e-06, "num_tokens": 1441807.0, "completions/mean_length": 43.125, "completions/min_length": 37.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.125, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.8190872669219971, "rewards/meter/std": 0.29186952114105225, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9653066396713257, "rewards/repeat_soft/std": 0.0251996461302042, "rewards/judge_quality/mean": 0.3474999666213989, "rewards/judge_quality/std": 0.11310551315546036, "rewards/total_composite/mean": 0.7193698883056641, "rewards/total_composite/std": 0.12337921559810638, "reward": 0.7193698883056641, "reward_std": 0.12337920814752579, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1880858987569809, "sampling/sampling_logp_difference/max": 1.782886266708374, "sampling/importance_sampling_ratio/min": 0.1681521236896515, "sampling/importance_sampling_ratio/mean": 1.024956464767456, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7280959486961365, "clip_ratio/low_mean": 0.05307610146701336, "clip_ratio/low_min": 0.05307610146701336, "clip_ratio/high_mean": 0.1322673549875617, "clip_ratio/high_max": 0.1322673549875617, "clip_ratio/region_mean": 0.18534345645457506, "reward_total_mean": 0.7193698883056641, "reward_meter_mean": 0.8190872669219971, "reward_meter_std": 0.29186952114105225, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9653066396713257, "reward_repeat_soft_std": 0.0251996461302042, "reward_judge_quality_mean": 0.3474999666213989, "reward_judge_quality_std": 0.11310551315546036, "reward_total_composite_mean": 0.7193698883056641, "reward_total_composite_std": 0.12337921559810638} {"timestamp_utc": "2026-04-13T00:10:25Z", "mode": "train", "global_step": 786, "epoch": 0.07895529884480161, "loss": -0.0541, "grad_norm": 3.806586742401123, "learning_rate": 7.621212121212122e-06, "num_tokens": 1443091.0, "completions/mean_length": 81.5, "completions/min_length": 18.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 20.0, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 22.0, "rewards/meter/mean": 0.5500002503395081, "rewards/meter/std": 0.47126874327659607, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9583790898323059, "rewards/repeat_soft/std": 0.011655605398118496, "rewards/judge_quality/mean": 0.35874998569488525, "rewards/judge_quality/std": 0.16225530207157135, "rewards/total_composite/mean": 0.5128167271614075, "rewards/total_composite/std": 0.27553120255470276, "reward": 0.5128167271614075, "reward_std": 0.27553120255470276, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1650397926568985, "sampling/sampling_logp_difference/max": 1.094635009765625, "sampling/importance_sampling_ratio/min": 0.33466172218322754, "sampling/importance_sampling_ratio/mean": 1.0402448177337646, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2430917918682098, "clip_ratio/low_mean": 0.04861111240461469, "clip_ratio/low_min": 0.04861111240461469, "clip_ratio/high_mean": 0.0433612447232008, "clip_ratio/high_max": 0.0433612447232008, "clip_ratio/region_mean": 0.09197235712781549, "reward_total_mean": 0.5128167271614075, "reward_meter_mean": 0.5500002503395081, "reward_meter_std": 0.47126874327659607, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9583790898323059, "reward_repeat_soft_std": 0.011655605398118496, "reward_judge_quality_mean": 0.35874998569488525, "reward_judge_quality_std": 0.16225530207157135, "reward_total_composite_mean": 0.5128167271614075, "reward_total_composite_std": 0.27553120255470276} {"timestamp_utc": "2026-04-13T00:10:38Z", "mode": "train", "global_step": 787, "epoch": 0.0790557508789553, "loss": -0.041, "grad_norm": 4.162282466888428, "learning_rate": 7.618181818181819e-06, "num_tokens": 1444659.0, "completions/mean_length": 97.0, "completions/min_length": 33.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 37.71428680419922, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.39725053310394287, "rewards/meter/std": 0.49465036392211914, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.955105721950531, "rewards/repeat_soft/std": 0.06175392493605614, "rewards/judge_quality/mean": 0.3387500047683716, "rewards/judge_quality/std": 0.20897282660007477, "rewards/total_composite/mean": 0.4844578504562378, "rewards/total_composite/std": 0.31594008207321167, "reward": 0.4844578504562378, "reward_std": 0.3159400522708893, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16112840175628662, "sampling/sampling_logp_difference/max": 1.456495761871338, "sampling/importance_sampling_ratio/min": 0.23305150866508484, "sampling/importance_sampling_ratio/mean": 1.0203828811645508, "sampling/importance_sampling_ratio/max": 1.7482166290283203, "entropy": 1.1995339393615723, "clip_ratio/low_mean": 0.08105473686009645, "clip_ratio/low_min": 0.08105473686009645, "clip_ratio/high_mean": 0.07209967449307442, "clip_ratio/high_max": 0.07209967449307442, "clip_ratio/region_mean": 0.15315441135317087, "reward_total_mean": 0.4844578504562378, "reward_meter_mean": 0.39725053310394287, "reward_meter_std": 0.49465036392211914, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.955105721950531, "reward_repeat_soft_std": 0.06175392493605614, "reward_judge_quality_mean": 0.3387500047683716, "reward_judge_quality_std": 0.20897282660007477, "reward_total_composite_mean": 0.4844578504562378, "reward_total_composite_std": 0.31594008207321167} {"timestamp_utc": "2026-04-13T00:10:49Z", "mode": "train", "global_step": 788, "epoch": 0.07915620291310899, "loss": 0.0154, "grad_norm": 12.917116165161133, "learning_rate": 7.6151515151515155e-06, "num_tokens": 1446354.0, "completions/mean_length": 41.875, "completions/min_length": 34.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.875, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9712004065513611, "rewards/meter/std": 0.03110371343791485, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8993959426879883, "rewards/repeat_soft/std": 0.08641824871301651, "rewards/judge_quality/mean": 0.36374998092651367, "rewards/judge_quality/std": 0.09500939399003983, "rewards/total_composite/mean": 0.7861047983169556, "rewards/total_composite/std": 0.033442266285419464, "reward": 0.7861047983169556, "reward_std": 0.03344227746129036, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14490124583244324, "sampling/sampling_logp_difference/max": 1.3387832641601562, "sampling/importance_sampling_ratio/min": 0.26216447353363037, "sampling/importance_sampling_ratio/mean": 1.0169587135314941, "sampling/importance_sampling_ratio/max": 1.823937177658081, "entropy": 1.2646863907575607, "clip_ratio/low_mean": 0.06490581668913364, "clip_ratio/low_min": 0.06490581668913364, "clip_ratio/high_mean": 0.08818126004189253, "clip_ratio/high_max": 0.08818126004189253, "clip_ratio/region_mean": 0.15308707673102617, "reward_total_mean": 0.7861047983169556, "reward_meter_mean": 0.9712004065513611, "reward_meter_std": 0.03110371343791485, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8993959426879883, "reward_repeat_soft_std": 0.08641824871301651, "reward_judge_quality_mean": 0.36374998092651367, "reward_judge_quality_std": 0.09500939399003983, "reward_total_composite_mean": 0.7861047983169556, "reward_total_composite_std": 0.033442266285419464} {"timestamp_utc": "2026-04-13T00:10:56Z", "mode": "train", "global_step": 789, "epoch": 0.07925665494726268, "loss": -0.0388, "grad_norm": 9.826400756835938, "learning_rate": 7.612121212121213e-06, "num_tokens": 1448145.0, "completions/mean_length": 50.875, "completions/min_length": 39.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.875, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.8449762463569641, "rewards/meter/std": 0.32970231771469116, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8584885597229004, "rewards/repeat_soft/std": 0.11816152930259705, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.08166787773370743, "rewards/total_composite/mean": 0.7304631471633911, "rewards/total_composite/std": 0.1456497609615326, "reward": 0.7304631471633911, "reward_std": 0.1456497460603714, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10181678831577301, "sampling/sampling_logp_difference/max": 1.1047344207763672, "sampling/importance_sampling_ratio/min": 0.3312988579273224, "sampling/importance_sampling_ratio/mean": 1.0197101831436157, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6203775219619274, "clip_ratio/low_mean": 0.04287277814000845, "clip_ratio/low_min": 0.04287277814000845, "clip_ratio/high_mean": 0.08102482836693525, "clip_ratio/high_max": 0.08102482836693525, "clip_ratio/region_mean": 0.1238976065069437, "reward_total_mean": 0.7304631471633911, "reward_meter_mean": 0.8449762463569641, "reward_meter_std": 0.32970231771469116, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8584885597229004, "reward_repeat_soft_std": 0.11816152930259705, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.08166787773370743, "reward_total_composite_mean": 0.7304631471633911, "reward_total_composite_std": 0.1456497609615326} {"timestamp_utc": "2026-04-13T00:11:09Z", "mode": "train", "global_step": 790, "epoch": 0.07935710698141638, "loss": -0.1919, "grad_norm": 1.7932337522506714, "learning_rate": 7.609090909090909e-06, "num_tokens": 1450662.0, "completions/mean_length": 158.625, "completions/min_length": 85.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 108.14286041259766, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.944664478302002, "rewards/meter/std": 0.11263131350278854, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.2182178944349289, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.5896071791648865, "rewards/repeat_soft/std": 0.21380387246608734, "rewards/judge_quality/mean": 0.1899999976158142, "rewards/judge_quality/std": 0.11501552164554596, "rewards/total_composite/mean": 0.5931099057197571, "rewards/total_composite/std": 0.25086307525634766, "reward": 0.5931099057197571, "reward_std": 0.25086304545402527, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0896620899438858, "sampling/sampling_logp_difference/max": 1.356989860534668, "sampling/importance_sampling_ratio/min": 0.25743454694747925, "sampling/importance_sampling_ratio/mean": 1.022432565689087, "sampling/importance_sampling_ratio/max": 1.8125594854354858, "entropy": 0.688418909907341, "clip_ratio/low_mean": 0.014583333395421505, "clip_ratio/low_min": 0.014583333395421505, "clip_ratio/high_mean": 0.0634547364898026, "clip_ratio/high_max": 0.0634547364898026, "clip_ratio/region_mean": 0.0780380698852241, "reward_total_mean": 0.5931099057197571, "reward_meter_mean": 0.944664478302002, "reward_meter_std": 0.11263131350278854, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.2182178944349289, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.5896071791648865, "reward_repeat_soft_std": 0.21380387246608734, "reward_judge_quality_mean": 0.1899999976158142, "reward_judge_quality_std": 0.11501552164554596, "reward_total_composite_mean": 0.5931099057197571, "reward_total_composite_std": 0.25086307525634766} {"timestamp_utc": "2026-04-13T00:11:27Z", "mode": "train", "global_step": 791, "epoch": 0.07945755901557007, "loss": -0.1242, "grad_norm": 2.884307622909546, "learning_rate": 7.606060606060606e-06, "num_tokens": 1452492.0, "completions/mean_length": 190.75, "completions/min_length": 66.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 83.66667175292969, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.7524266242980957, "rewards/meter/std": 0.3054893910884857, "rewards/count_adherence/mean": 0.71875, "rewards/count_adherence/std": 0.31160587072372437, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8094086050987244, "rewards/repeat_soft/std": 0.15367022156715393, "rewards/judge_quality/mean": 0.1862500011920929, "rewards/judge_quality/std": 0.11697832494974136, "rewards/total_composite/mean": 0.583220362663269, "rewards/total_composite/std": 0.20262089371681213, "reward": 0.583220362663269, "reward_std": 0.20262089371681213, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.156460240483284, "sampling/sampling_logp_difference/max": 1.5259346961975098, "sampling/importance_sampling_ratio/min": 0.21741774678230286, "sampling/importance_sampling_ratio/mean": 1.0211237668991089, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0052568390965462, "clip_ratio/low_mean": 0.015625, "clip_ratio/low_min": 0.015625, "clip_ratio/high_mean": 0.07915621809661388, "clip_ratio/high_max": 0.07915621809661388, "clip_ratio/region_mean": 0.09478121809661388, "reward_total_mean": 0.583220362663269, "reward_meter_mean": 0.7524266242980957, "reward_meter_std": 0.3054893910884857, "reward_count_adherence_mean": 0.71875, "reward_count_adherence_std": 0.31160587072372437, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8094086050987244, "reward_repeat_soft_std": 0.15367022156715393, "reward_judge_quality_mean": 0.1862500011920929, "reward_judge_quality_std": 0.11697832494974136, "reward_total_composite_mean": 0.583220362663269, "reward_total_composite_std": 0.20262089371681213} {"timestamp_utc": "2026-04-13T00:11:34Z", "mode": "train", "global_step": 792, "epoch": 0.07955801104972375, "loss": 0.0263, "grad_norm": 25.107107162475586, "learning_rate": 7.603030303030303e-06, "num_tokens": 1453991.0, "completions/mean_length": 19.375, "completions/min_length": 17.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.375, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.74037766456604, "rewards/meter/std": 0.4404917359352112, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9596153497695923, "rewards/repeat_soft/std": 0.008158913813531399, "rewards/judge_quality/mean": 0.8650000095367432, "rewards/judge_quality/std": 0.18031719326972961, "rewards/total_composite/mean": 0.8386315107345581, "rewards/total_composite/std": 0.1967504471540451, "reward": 0.8386315107345581, "reward_std": 0.1967504471540451, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13281717896461487, "sampling/sampling_logp_difference/max": 0.8120079040527344, "sampling/importance_sampling_ratio/min": 0.44396573305130005, "sampling/importance_sampling_ratio/mean": 1.0091503858566284, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.991125650703907, "clip_ratio/low_mean": 0.03976715775206685, "clip_ratio/low_min": 0.03976715775206685, "clip_ratio/high_mean": 0.09391872910782695, "clip_ratio/high_max": 0.09391872910782695, "clip_ratio/region_mean": 0.1336858868598938, "reward_total_mean": 0.8386315107345581, "reward_meter_mean": 0.74037766456604, "reward_meter_std": 0.4404917359352112, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9596153497695923, "reward_repeat_soft_std": 0.008158913813531399, "reward_judge_quality_mean": 0.8650000095367432, "reward_judge_quality_std": 0.18031719326972961, "reward_total_composite_mean": 0.8386315107345581, "reward_total_composite_std": 0.1967504471540451} {"timestamp_utc": "2026-04-13T00:11:42Z", "mode": "train", "global_step": 793, "epoch": 0.07965846308387745, "loss": 0.1726, "grad_norm": 18.71817970275879, "learning_rate": 7.600000000000001e-06, "num_tokens": 1455470.0, "completions/mean_length": 32.875, "completions/min_length": 24.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.875, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.6676452159881592, "rewards/meter/std": 0.3526809811592102, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9748848676681519, "rewards/repeat_soft/std": 0.028298022225499153, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.20860078930854797, "rewards/total_composite/mean": 0.6799288392066956, "rewards/total_composite/std": 0.17561253905296326, "reward": 0.6799288392066956, "reward_std": 0.17561253905296326, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2015603631734848, "sampling/sampling_logp_difference/max": 1.600982666015625, "sampling/importance_sampling_ratio/min": 0.2016982138156891, "sampling/importance_sampling_ratio/mean": 1.033434510231018, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.585381381213665, "clip_ratio/low_mean": 0.05899256281554699, "clip_ratio/low_min": 0.05899256281554699, "clip_ratio/high_mean": 0.12553749606013298, "clip_ratio/high_max": 0.12553749606013298, "clip_ratio/region_mean": 0.18453005887567997, "reward_total_mean": 0.6799288392066956, "reward_meter_mean": 0.6676452159881592, "reward_meter_std": 0.3526809811592102, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9748848676681519, "reward_repeat_soft_std": 0.028298022225499153, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.20860078930854797, "reward_total_composite_mean": 0.6799288392066956, "reward_total_composite_std": 0.17561253905296326} {"timestamp_utc": "2026-04-13T00:11:55Z", "mode": "train", "global_step": 794, "epoch": 0.07975891511803114, "loss": -0.1236, "grad_norm": 1.7425907850265503, "learning_rate": 7.596969696969697e-06, "num_tokens": 1457205.0, "completions/mean_length": 230.875, "completions/min_length": 55.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 62.20000076293945, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.8339755535125732, "rewards/meter/std": 0.29440852999687195, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.22160132229328156, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.8414033651351929, "rewards/repeat_soft/std": 0.15028606355190277, "rewards/judge_quality/mean": 0.13125000894069672, "rewards/judge_quality/std": 0.09234059602022171, "rewards/total_composite/mean": 0.43406930565834045, "rewards/total_composite/std": 0.3600548803806305, "reward": 0.43406930565834045, "reward_std": 0.3600548505783081, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1324991136789322, "sampling/sampling_logp_difference/max": 1.3643569946289062, "sampling/importance_sampling_ratio/min": 0.2555449604988098, "sampling/importance_sampling_ratio/mean": 1.0244596004486084, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8110391795635223, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10324553772807121, "clip_ratio/high_max": 0.10324553772807121, "clip_ratio/region_mean": 0.10324553772807121, "reward_total_mean": 0.43406930565834045, "reward_meter_mean": 0.8339755535125732, "reward_meter_std": 0.29440852999687195, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.22160132229328156, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.8414033651351929, "reward_repeat_soft_std": 0.15028606355190277, "reward_judge_quality_mean": 0.13125000894069672, "reward_judge_quality_std": 0.09234059602022171, "reward_total_composite_mean": 0.43406930565834045, "reward_total_composite_std": 0.3600548803806305} {"timestamp_utc": "2026-04-13T00:12:08Z", "mode": "train", "global_step": 795, "epoch": 0.07985936715218483, "loss": -0.1553, "grad_norm": 1.964593529701233, "learning_rate": 7.593939393939395e-06, "num_tokens": 1459155.0, "completions/mean_length": 118.75, "completions/min_length": 53.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 62.57143020629883, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9706916809082031, "rewards/meter/std": 0.055730897933244705, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7558030486106873, "rewards/repeat_soft/std": 0.2071211189031601, "rewards/judge_quality/mean": 0.19625000655651093, "rewards/judge_quality/std": 0.11070391535758972, "rewards/total_composite/mean": 0.6090935468673706, "rewards/total_composite/std": 0.249989852309227, "reward": 0.6090935468673706, "reward_std": 0.249989852309227, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13254427909851074, "sampling/sampling_logp_difference/max": 1.4981393814086914, "sampling/importance_sampling_ratio/min": 0.22354571521282196, "sampling/importance_sampling_ratio/mean": 1.0085581541061401, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9366035610437393, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.13668652903288603, "clip_ratio/high_max": 0.13668652903288603, "clip_ratio/region_mean": 0.13668652903288603, "reward_total_mean": 0.6090935468673706, "reward_meter_mean": 0.9706916809082031, "reward_meter_std": 0.055730897933244705, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7558030486106873, "reward_repeat_soft_std": 0.2071211189031601, "reward_judge_quality_mean": 0.19625000655651093, "reward_judge_quality_std": 0.11070391535758972, "reward_total_composite_mean": 0.6090935468673706, "reward_total_composite_std": 0.249989852309227} {"timestamp_utc": "2026-04-13T00:12:15Z", "mode": "train", "global_step": 796, "epoch": 0.07995981918633853, "loss": 0.0611, "grad_norm": 14.796483993530273, "learning_rate": 7.590909090909091e-06, "num_tokens": 1460749.0, "completions/mean_length": 43.25, "completions/min_length": 34.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.25, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.806858241558075, "rewards/meter/std": 0.31579187512397766, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9852569699287415, "rewards/repeat_soft/std": 0.023008430376648903, "rewards/judge_quality/mean": 0.6475000381469727, "rewards/judge_quality/std": 0.30742016434669495, "rewards/total_composite/mean": 0.7664731740951538, "rewards/total_composite/std": 0.3185473680496216, "reward": 0.7664731740951538, "reward_std": 0.3185473680496216, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1622677892446518, "sampling/sampling_logp_difference/max": 1.0784215927124023, "sampling/importance_sampling_ratio/min": 0.3401319682598114, "sampling/importance_sampling_ratio/mean": 1.018595576286316, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.266797423362732, "clip_ratio/low_mean": 0.025510204955935478, "clip_ratio/low_min": 0.025510204955935478, "clip_ratio/high_mean": 0.12554714735597372, "clip_ratio/high_max": 0.12554714735597372, "clip_ratio/region_mean": 0.1510573523119092, "reward_total_mean": 0.7664731740951538, "reward_meter_mean": 0.806858241558075, "reward_meter_std": 0.31579187512397766, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9852569699287415, "reward_repeat_soft_std": 0.023008430376648903, "reward_judge_quality_mean": 0.6475000381469727, "reward_judge_quality_std": 0.30742016434669495, "reward_total_composite_mean": 0.7664731740951538, "reward_total_composite_std": 0.3185473680496216} {"timestamp_utc": "2026-04-13T00:12:22Z", "mode": "train", "global_step": 797, "epoch": 0.08006027122049221, "loss": -0.0128, "grad_norm": 14.334541320800781, "learning_rate": 7.587878787878788e-06, "num_tokens": 1462560.0, "completions/mean_length": 54.375, "completions/min_length": 46.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.375, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.8016561269760132, "rewards/meter/std": 0.32069116830825806, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8828722834587097, "rewards/repeat_soft/std": 0.07804501056671143, "rewards/judge_quality/mean": 0.30124998092651367, "rewards/judge_quality/std": 0.10398317128419876, "rewards/total_composite/mean": 0.6519075036048889, "rewards/total_composite/std": 0.15880799293518066, "reward": 0.6519075036048889, "reward_std": 0.15880799293518066, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16211773455142975, "sampling/sampling_logp_difference/max": 2.2337594032287598, "sampling/importance_sampling_ratio/min": 0.1071249395608902, "sampling/importance_sampling_ratio/mean": 1.026573896408081, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1626599729061127, "clip_ratio/low_mean": 0.03008199855685234, "clip_ratio/low_min": 0.03008199855685234, "clip_ratio/high_mean": 0.1264455933123827, "clip_ratio/high_max": 0.1264455933123827, "clip_ratio/region_mean": 0.15652759186923504, "reward_total_mean": 0.6519075036048889, "reward_meter_mean": 0.8016561269760132, "reward_meter_std": 0.32069116830825806, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8828722834587097, "reward_repeat_soft_std": 0.07804501056671143, "reward_judge_quality_mean": 0.30124998092651367, "reward_judge_quality_std": 0.10398317128419876, "reward_total_composite_mean": 0.6519075036048889, "reward_total_composite_std": 0.15880799293518066} {"timestamp_utc": "2026-04-13T00:12:28Z", "mode": "train", "global_step": 798, "epoch": 0.0801607232546459, "loss": 0.0299, "grad_norm": 15.187627792358398, "learning_rate": 7.584848484848486e-06, "num_tokens": 1464355.0, "completions/mean_length": 43.375, "completions/min_length": 36.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.375, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.6393726468086243, "rewards/meter/std": 0.37396690249443054, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9682679176330566, "rewards/repeat_soft/std": 0.03498344495892525, "rewards/judge_quality/mean": 0.35999998450279236, "rewards/judge_quality/std": 0.09165150672197342, "rewards/total_composite/mean": 0.579789400100708, "rewards/total_composite/std": 0.2602539658546448, "reward": 0.579789400100708, "reward_std": 0.2602539658546448, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18980687856674194, "sampling/sampling_logp_difference/max": 2.9651105403900146, "sampling/importance_sampling_ratio/min": 0.05155477300286293, "sampling/importance_sampling_ratio/mean": 1.0140459537506104, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9441961199045181, "clip_ratio/low_mean": 0.04679679684340954, "clip_ratio/low_min": 0.04679679684340954, "clip_ratio/high_mean": 0.10688188206404448, "clip_ratio/high_max": 0.10688188206404448, "clip_ratio/region_mean": 0.153678678907454, "reward_total_mean": 0.579789400100708, "reward_meter_mean": 0.6393726468086243, "reward_meter_std": 0.37396690249443054, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9682679176330566, "reward_repeat_soft_std": 0.03498344495892525, "reward_judge_quality_mean": 0.35999998450279236, "reward_judge_quality_std": 0.09165150672197342, "reward_total_composite_mean": 0.579789400100708, "reward_total_composite_std": 0.2602539658546448} {"timestamp_utc": "2026-04-13T00:12:40Z", "mode": "train", "global_step": 799, "epoch": 0.0802611752887996, "loss": -0.1048, "grad_norm": 5.087594985961914, "learning_rate": 7.581818181818183e-06, "num_tokens": 1466013.0, "completions/mean_length": 104.25, "completions/min_length": 30.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 46.000003814697266, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.44222718477249146, "rewards/meter/std": 0.3267405033111572, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9225562810897827, "rewards/repeat_soft/std": 0.16239245235919952, "rewards/judge_quality/mean": 0.36374998092651367, "rewards/judge_quality/std": 0.14302222430706024, "rewards/total_composite/mean": 0.4397534728050232, "rewards/total_composite/std": 0.30961793661117554, "reward": 0.4397534728050232, "reward_std": 0.30961793661117554, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19972014427185059, "sampling/sampling_logp_difference/max": 1.647073745727539, "sampling/importance_sampling_ratio/min": 0.19261270761489868, "sampling/importance_sampling_ratio/mean": 0.9930789470672607, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8867237716913223, "clip_ratio/low_mean": 0.09724284149706364, "clip_ratio/low_min": 0.09724284149706364, "clip_ratio/high_mean": 0.09722114354372025, "clip_ratio/high_max": 0.09722114354372025, "clip_ratio/region_mean": 0.19446398504078388, "reward_total_mean": 0.4397534728050232, "reward_meter_mean": 0.44222718477249146, "reward_meter_std": 0.3267405033111572, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9225562810897827, "reward_repeat_soft_std": 0.16239245235919952, "reward_judge_quality_mean": 0.36374998092651367, "reward_judge_quality_std": 0.14302222430706024, "reward_total_composite_mean": 0.4397534728050232, "reward_total_composite_std": 0.30961793661117554} {"timestamp_utc": "2026-04-13T00:12:52Z", "mode": "train", "global_step": 800, "epoch": 0.0803616273229533, "loss": -0.0796, "grad_norm": 1.7772691249847412, "learning_rate": 7.57878787878788e-06, "num_tokens": 1467613.0, "completions/mean_length": 220.0, "completions/min_length": 40.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 44.79999923706055, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.5926225185394287, "rewards/meter/std": 0.4038395881652832, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9480060338973999, "rewards/repeat_soft/std": 0.056003496050834656, "rewards/judge_quality/mean": 0.17500001192092896, "rewards/judge_quality/std": 0.13887301087379456, "rewards/total_composite/mean": 0.4426027834415436, "rewards/total_composite/std": 0.3128497302532196, "reward": 0.4426027834415436, "reward_std": 0.3128497302532196, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15526747703552246, "sampling/sampling_logp_difference/max": 1.5290813446044922, "sampling/importance_sampling_ratio/min": 0.21673469245433807, "sampling/importance_sampling_ratio/mean": 1.0455939769744873, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0655998587608337, "clip_ratio/low_mean": 0.02616279013454914, "clip_ratio/low_min": 0.02616279013454914, "clip_ratio/high_mean": 0.07716503459960222, "clip_ratio/high_max": 0.07716503459960222, "clip_ratio/region_mean": 0.10332782473415136, "reward_total_mean": 0.4426027834415436, "reward_meter_mean": 0.5926225185394287, "reward_meter_std": 0.4038395881652832, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9480060338973999, "reward_repeat_soft_std": 0.056003496050834656, "reward_judge_quality_mean": 0.17500001192092896, "reward_judge_quality_std": 0.13887301087379456, "reward_total_composite_mean": 0.4426027834415436, "reward_total_composite_std": 0.3128497302532196} {"timestamp_utc": "2026-04-13T00:14:17Z", "mode": "eval", "global_step": 800, "epoch": 0.0803616273229533, "eval_loss": NaN, "eval_runtime": 84.745, "eval_samples_per_second": 0.944, "eval_steps_per_second": 0.118, "eval_num_tokens": 1467613.0, "eval_completions/mean_length": 117.5375, "eval_completions/min_length": 27.8, "eval_completions/max_length": 427.2, "eval_completions/clipped_ratio": 0.1375, "eval_completions/mean_terminated_length": 54.66023979187012, "eval_completions/min_terminated_length": 27.8, "eval_completions/max_terminated_length": 96.9, "eval_rewards/meter/mean": 0.673719871044159, "eval_rewards/meter/std": 0.35566111356019975, "eval_rewards/count_adherence/mean": 0.8635416626930237, "eval_rewards/count_adherence/std": 0.17047111690044403, "eval_rewards/hard_gate/mean": 0.875, "eval_rewards/hard_gate/std": 0.2748226195573807, "eval_rewards/repeat_soft/mean": 0.8381768941879273, "eval_rewards/repeat_soft/std": 0.15132683962583543, "eval_rewards/judge_quality/mean": 0.2958749935030937, "eval_rewards/judge_quality/std": 0.18107932135462762, "eval_rewards/total_composite/mean": 0.5625116974115372, "eval_rewards/total_composite/std": 0.24842821806669235, "eval_reward": 0.5625116974115372, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.07896818667650223, "eval_sampling/sampling_logp_difference/max": 0.959096097946167, "eval_sampling/importance_sampling_ratio/min": 0.3925798863172531, "eval_sampling/importance_sampling_ratio/mean": 1.0213791131973267, "eval_sampling/importance_sampling_ratio/max": 1.5346822619438172, "eval_entropy": 0.9235897898674011, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5625116974115372, "eval_reward_meter_mean": 0.673719871044159, "eval_reward_meter_std": 0.35566111356019975, "eval_reward_count_adherence_mean": 0.8635416626930237, "eval_reward_count_adherence_std": 0.17047111690044403, "eval_reward_hard_gate_mean": 0.875, "eval_reward_hard_gate_std": 0.2748226195573807, "eval_reward_repeat_soft_mean": 0.8381768941879273, "eval_reward_repeat_soft_std": 0.15132683962583543, "eval_reward_judge_quality_mean": 0.2958749935030937, "eval_reward_judge_quality_std": 0.18107932135462762, "eval_reward_total_composite_mean": 0.5625116974115372, "eval_reward_total_composite_std": 0.24842821806669235} {"timestamp_utc": "2026-04-13T00:14:33Z", "mode": "train", "global_step": 801, "epoch": 0.08046207935710697, "loss": -0.0927, "grad_norm": 3.6296565532684326, "learning_rate": 7.5757575757575764e-06, "num_tokens": 1469137.0, "completions/mean_length": 94.5, "completions/min_length": 31.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 34.85714340209961, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.509043276309967, "rewards/meter/std": 0.413009911775589, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8842312097549438, "rewards/repeat_soft/std": 0.11565447598695755, "rewards/judge_quality/mean": 0.3137499988079071, "rewards/judge_quality/std": 0.20570003986358643, "rewards/total_composite/mean": 0.5299092531204224, "rewards/total_composite/std": 0.27928170561790466, "reward": 0.5299092531204224, "reward_std": 0.27928170561790466, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13712213933467865, "sampling/sampling_logp_difference/max": 1.1595726013183594, "sampling/importance_sampling_ratio/min": 0.3136201798915863, "sampling/importance_sampling_ratio/mean": 1.0287715196609497, "sampling/importance_sampling_ratio/max": 1.9268735647201538, "entropy": 1.1218813136219978, "clip_ratio/low_mean": 0.05230015702545643, "clip_ratio/low_min": 0.05230015702545643, "clip_ratio/high_mean": 0.08274779375642538, "clip_ratio/high_max": 0.08274779375642538, "clip_ratio/region_mean": 0.1350479507818818, "reward_total_mean": 0.5299092531204224, "reward_meter_mean": 0.509043276309967, "reward_meter_std": 0.413009911775589, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8842312097549438, "reward_repeat_soft_std": 0.11565447598695755, "reward_judge_quality_mean": 0.3137499988079071, "reward_judge_quality_std": 0.20570003986358643, "reward_total_composite_mean": 0.5299092531204224, "reward_total_composite_std": 0.27928170561790466} {"timestamp_utc": "2026-04-13T00:14:41Z", "mode": "train", "global_step": 802, "epoch": 0.08056253139126067, "loss": 0.0302, "grad_norm": 14.719854354858398, "learning_rate": 7.572727272727274e-06, "num_tokens": 1471013.0, "completions/mean_length": 50.5, "completions/min_length": 37.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.5, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.8768517971038818, "rewards/meter/std": 0.29854634404182434, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9245647192001343, "rewards/repeat_soft/std": 0.06408214569091797, "rewards/judge_quality/mean": 0.3050000071525574, "rewards/judge_quality/std": 0.14322808384895325, "rewards/total_composite/mean": 0.7191647291183472, "rewards/total_composite/std": 0.16042527556419373, "reward": 0.7191647291183472, "reward_std": 0.16042527556419373, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1327352374792099, "sampling/sampling_logp_difference/max": 2.1797409057617188, "sampling/importance_sampling_ratio/min": 0.11307081580162048, "sampling/importance_sampling_ratio/mean": 1.0079870223999023, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9343869835138321, "clip_ratio/low_mean": 0.0273583447560668, "clip_ratio/low_min": 0.0273583447560668, "clip_ratio/high_mean": 0.1259198933839798, "clip_ratio/high_max": 0.1259198933839798, "clip_ratio/region_mean": 0.1532782381400466, "reward_total_mean": 0.7191647291183472, "reward_meter_mean": 0.8768517971038818, "reward_meter_std": 0.29854634404182434, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9245647192001343, "reward_repeat_soft_std": 0.06408214569091797, "reward_judge_quality_mean": 0.3050000071525574, "reward_judge_quality_std": 0.14322808384895325, "reward_total_composite_mean": 0.7191647291183472, "reward_total_composite_std": 0.16042527556419373} {"timestamp_utc": "2026-04-13T00:14:50Z", "mode": "train", "global_step": 803, "epoch": 0.08066298342541436, "loss": 0.0526, "grad_norm": 17.17169952392578, "learning_rate": 7.56969696969697e-06, "num_tokens": 1472705.0, "completions/mean_length": 42.5, "completions/min_length": 26.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.5, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.20047828555107117, "rewards/meter/std": 0.28125572204589844, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.17817415297031403, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9579493999481201, "rewards/repeat_soft/std": 0.050228312611579895, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.46701017022132874, "rewards/total_composite/std": 0.15982644259929657, "reward": 0.46701017022132874, "reward_std": 0.15982644259929657, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1449030041694641, "sampling/sampling_logp_difference/max": 2.2311935424804688, "sampling/importance_sampling_ratio/min": 0.10740016400814056, "sampling/importance_sampling_ratio/mean": 1.0055378675460815, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8126713410019875, "clip_ratio/low_mean": 0.06435279920697212, "clip_ratio/low_min": 0.06435279920697212, "clip_ratio/high_mean": 0.05750161409378052, "clip_ratio/high_max": 0.05750161409378052, "clip_ratio/region_mean": 0.12185441330075264, "reward_total_mean": 0.46701017022132874, "reward_meter_mean": 0.20047828555107117, "reward_meter_std": 0.28125572204589844, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.17817415297031403, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9579493999481201, "reward_repeat_soft_std": 0.050228312611579895, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.46701017022132874, "reward_total_composite_std": 0.15982644259929657} {"timestamp_utc": "2026-04-13T00:14:58Z", "mode": "train", "global_step": 804, "epoch": 0.08076343545956806, "loss": 0.0311, "grad_norm": 12.20918083190918, "learning_rate": 7.566666666666667e-06, "num_tokens": 1474869.0, "completions/mean_length": 78.5, "completions/min_length": 66.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.5, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.9159865975379944, "rewards/meter/std": 0.12517111003398895, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.5512501001358032, "rewards/repeat_soft/std": 0.21963344514369965, "rewards/judge_quality/mean": 0.22374999523162842, "rewards/judge_quality/std": 0.12805551290512085, "rewards/total_composite/mean": 0.6544439792633057, "rewards/total_composite/std": 0.07384175807237625, "reward": 0.6544439792633057, "reward_std": 0.07384176552295685, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10944996774196625, "sampling/sampling_logp_difference/max": 1.3579351902008057, "sampling/importance_sampling_ratio/min": 0.2571912705898285, "sampling/importance_sampling_ratio/mean": 1.0107202529907227, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6957814693450928, "clip_ratio/low_mean": 0.05701480666175485, "clip_ratio/low_min": 0.05701480666175485, "clip_ratio/high_mean": 0.039789898321032524, "clip_ratio/high_max": 0.039789898321032524, "clip_ratio/region_mean": 0.09680470498278737, "reward_total_mean": 0.6544439792633057, "reward_meter_mean": 0.9159865975379944, "reward_meter_std": 0.12517111003398895, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.5512501001358032, "reward_repeat_soft_std": 0.21963344514369965, "reward_judge_quality_mean": 0.22374999523162842, "reward_judge_quality_std": 0.12805551290512085, "reward_total_composite_mean": 0.6544439792633057, "reward_total_composite_std": 0.07384175807237625} {"timestamp_utc": "2026-04-13T00:15:05Z", "mode": "train", "global_step": 805, "epoch": 0.08086388749372175, "loss": 0.0359, "grad_norm": 13.51599407196045, "learning_rate": 7.563636363636364e-06, "num_tokens": 1476545.0, "completions/mean_length": 40.5, "completions/min_length": 32.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.5, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.6997767686843872, "rewards/meter/std": 0.4158882796764374, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9238604307174683, "rewards/repeat_soft/std": 0.046184197068214417, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.6791605949401855, "rewards/total_composite/std": 0.18664810061454773, "reward": 0.6791605949401855, "reward_std": 0.18664810061454773, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14606845378875732, "sampling/sampling_logp_difference/max": 1.5691699981689453, "sampling/importance_sampling_ratio/min": 0.20821791887283325, "sampling/importance_sampling_ratio/mean": 1.0057451725006104, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.851159855723381, "clip_ratio/low_mean": 0.04224738758057356, "clip_ratio/low_min": 0.04224738758057356, "clip_ratio/high_mean": 0.10966979712247849, "clip_ratio/high_max": 0.10966979712247849, "clip_ratio/region_mean": 0.15191718470305204, "reward_total_mean": 0.6791605949401855, "reward_meter_mean": 0.6997767686843872, "reward_meter_std": 0.4158882796764374, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9238604307174683, "reward_repeat_soft_std": 0.046184197068214417, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.6791605949401855, "reward_total_composite_std": 0.18664810061454773} {"timestamp_utc": "2026-04-13T00:15:19Z", "mode": "train", "global_step": 806, "epoch": 0.08096433952787543, "loss": -0.131, "grad_norm": 1.6838499307632446, "learning_rate": 7.560606060606062e-06, "num_tokens": 1478144.0, "completions/mean_length": 102.875, "completions/min_length": 36.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 44.42857360839844, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9895227551460266, "rewards/meter/std": 0.005883883219212294, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.856874942779541, "rewards/repeat_soft/std": 0.10524369776248932, "rewards/judge_quality/mean": 0.19374999403953552, "rewards/judge_quality/std": 0.1237436830997467, "rewards/total_composite/mean": 0.6507579684257507, "rewards/total_composite/std": 0.2655637860298157, "reward": 0.6507579684257507, "reward_std": 0.2655637562274933, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1636924296617508, "sampling/sampling_logp_difference/max": 2.0701560974121094, "sampling/importance_sampling_ratio/min": 0.12616609036922455, "sampling/importance_sampling_ratio/mean": 1.0317479372024536, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3520750403404236, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.12607576698064804, "clip_ratio/high_max": 0.12607576698064804, "clip_ratio/region_mean": 0.12607576698064804, "reward_total_mean": 0.6507579684257507, "reward_meter_mean": 0.9895227551460266, "reward_meter_std": 0.005883883219212294, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.856874942779541, "reward_repeat_soft_std": 0.10524369776248932, "reward_judge_quality_mean": 0.19374999403953552, "reward_judge_quality_std": 0.1237436830997467, "reward_total_composite_mean": 0.6507579684257507, "reward_total_composite_std": 0.2655637860298157} {"timestamp_utc": "2026-04-13T00:15:33Z", "mode": "train", "global_step": 807, "epoch": 0.08106479156202913, "loss": -0.0842, "grad_norm": 1.8313453197479248, "learning_rate": 7.557575757575758e-06, "num_tokens": 1479585.0, "completions/mean_length": 153.125, "completions/min_length": 27.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 33.5, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.6314228773117065, "rewards/meter/std": 0.39097195863723755, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9699854850769043, "rewards/repeat_soft/std": 0.036621496081352234, "rewards/judge_quality/mean": 0.3137500286102295, "rewards/judge_quality/std": 0.17492344975471497, "rewards/total_composite/mean": 0.5806738138198853, "rewards/total_composite/std": 0.304672509431839, "reward": 0.5806738138198853, "reward_std": 0.304672509431839, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15564914047718048, "sampling/sampling_logp_difference/max": 1.3370566368103027, "sampling/importance_sampling_ratio/min": 0.2626175284385681, "sampling/importance_sampling_ratio/mean": 1.0071018934249878, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7935311570763588, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10940734064206481, "clip_ratio/high_max": 0.10940734064206481, "clip_ratio/region_mean": 0.10940734064206481, "reward_total_mean": 0.5806738138198853, "reward_meter_mean": 0.6314228773117065, "reward_meter_std": 0.39097195863723755, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9699854850769043, "reward_repeat_soft_std": 0.036621496081352234, "reward_judge_quality_mean": 0.3137500286102295, "reward_judge_quality_std": 0.17492344975471497, "reward_total_composite_mean": 0.5806738138198853, "reward_total_composite_std": 0.304672509431839} {"timestamp_utc": "2026-04-13T00:15:45Z", "mode": "train", "global_step": 808, "epoch": 0.08116524359618282, "loss": -0.1211, "grad_norm": 2.162411689758301, "learning_rate": 7.5545454545454555e-06, "num_tokens": 1481131.0, "completions/mean_length": 175.25, "completions/min_length": 49.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.5694653987884521, "rewards/meter/std": 0.44741514325141907, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.3720119297504425, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9357702136039734, "rewards/repeat_soft/std": 0.037184540182352066, "rewards/judge_quality/mean": 0.3174999952316284, "rewards/judge_quality/std": 0.17782413959503174, "rewards/total_composite/mean": 0.5028548836708069, "rewards/total_composite/std": 0.34891432523727417, "reward": 0.5028548836708069, "reward_std": 0.34891432523727417, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11479105800390244, "sampling/sampling_logp_difference/max": 1.0630028247833252, "sampling/importance_sampling_ratio/min": 0.3454170227050781, "sampling/importance_sampling_ratio/mean": 0.9987236261367798, "sampling/importance_sampling_ratio/max": 1.9299747943878174, "entropy": 0.6774219274520874, "clip_ratio/low_mean": 0.0086805559694767, "clip_ratio/low_min": 0.0086805559694767, "clip_ratio/high_mean": 0.09449565224349499, "clip_ratio/high_max": 0.09449565224349499, "clip_ratio/region_mean": 0.10317620821297169, "reward_total_mean": 0.5028548836708069, "reward_meter_mean": 0.5694653987884521, "reward_meter_std": 0.44741514325141907, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.3720119297504425, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9357702136039734, "reward_repeat_soft_std": 0.037184540182352066, "reward_judge_quality_mean": 0.3174999952316284, "reward_judge_quality_std": 0.17782413959503174, "reward_total_composite_mean": 0.5028548836708069, "reward_total_composite_std": 0.34891432523727417} {"timestamp_utc": "2026-04-13T00:15:52Z", "mode": "train", "global_step": 809, "epoch": 0.08126569563033652, "loss": 0.0675, "grad_norm": 23.923175811767578, "learning_rate": 7.551515151515152e-06, "num_tokens": 1482685.0, "completions/mean_length": 32.25, "completions/min_length": 29.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.25, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9852284789085388, "rewards/meter/std": 0.010701124556362629, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7984150648117065, "rewards/repeat_soft/std": 0.08569476753473282, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.0843462198972702, "rewards/total_composite/mean": 0.6884474754333496, "rewards/total_composite/std": 0.2788808047771454, "reward": 0.6884474754333496, "reward_std": 0.2788808047771454, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13869169354438782, "sampling/sampling_logp_difference/max": 3.4885799884796143, "sampling/importance_sampling_ratio/min": 0.030544213950634003, "sampling/importance_sampling_ratio/mean": 1.0216662883758545, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6976450830698013, "clip_ratio/low_mean": 0.0071428571827709675, "clip_ratio/low_min": 0.0071428571827709675, "clip_ratio/high_mean": 0.09158328478224576, "clip_ratio/high_max": 0.09158328478224576, "clip_ratio/region_mean": 0.09872614196501672, "reward_total_mean": 0.6884474754333496, "reward_meter_mean": 0.9852284789085388, "reward_meter_std": 0.010701124556362629, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7984150648117065, "reward_repeat_soft_std": 0.08569476753473282, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.0843462198972702, "reward_total_composite_mean": 0.6884474754333496, "reward_total_composite_std": 0.2788808047771454} {"timestamp_utc": "2026-04-13T00:15:58Z", "mode": "train", "global_step": 810, "epoch": 0.08136614766449021, "loss": 0.0509, "grad_norm": 23.675704956054688, "learning_rate": 7.548484848484849e-06, "num_tokens": 1484149.0, "completions/mean_length": 21.0, "completions/min_length": 18.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.0, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.9052899479866028, "rewards/meter/std": 0.22368668019771576, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9311120510101318, "rewards/repeat_soft/std": 0.037920016795396805, "rewards/judge_quality/mean": 0.3962499797344208, "rewards/judge_quality/std": 0.09085899591445923, "rewards/total_composite/mean": 0.7693666815757751, "rewards/total_composite/std": 0.12343854457139969, "reward": 0.7693666815757751, "reward_std": 0.12343855202198029, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1422675997018814, "sampling/sampling_logp_difference/max": 1.0976552963256836, "sampling/importance_sampling_ratio/min": 0.33365246653556824, "sampling/importance_sampling_ratio/mean": 1.005897045135498, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0120654851198196, "clip_ratio/low_mean": 0.01923076994717121, "clip_ratio/low_min": 0.01923076994717121, "clip_ratio/high_mean": 0.1256605521775782, "clip_ratio/high_max": 0.1256605521775782, "clip_ratio/region_mean": 0.14489132212474942, "reward_total_mean": 0.7693666815757751, "reward_meter_mean": 0.9052899479866028, "reward_meter_std": 0.22368668019771576, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9311120510101318, "reward_repeat_soft_std": 0.037920016795396805, "reward_judge_quality_mean": 0.3962499797344208, "reward_judge_quality_std": 0.09085899591445923, "reward_total_composite_mean": 0.7693666815757751, "reward_total_composite_std": 0.12343854457139969} {"timestamp_utc": "2026-04-13T00:16:05Z", "mode": "train", "global_step": 811, "epoch": 0.0814665996986439, "loss": 0.1179, "grad_norm": 25.506473541259766, "learning_rate": 7.545454545454546e-06, "num_tokens": 1485797.0, "completions/mean_length": 31.0, "completions/min_length": 22.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.0, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.5983876585960388, "rewards/meter/std": 0.34947502613067627, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8755115270614624, "rewards/repeat_soft/std": 0.10106371343135834, "rewards/judge_quality/mean": 0.24250000715255737, "rewards/judge_quality/std": 0.11792854964733124, "rewards/total_composite/mean": 0.5795755982398987, "rewards/total_composite/std": 0.1329527199268341, "reward": 0.5795755982398987, "reward_std": 0.1329527199268341, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17879511415958405, "sampling/sampling_logp_difference/max": 3.2989840507507324, "sampling/importance_sampling_ratio/min": 0.03692065551877022, "sampling/importance_sampling_ratio/mean": 1.0272420644760132, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8765083998441696, "clip_ratio/low_mean": 0.05964052490890026, "clip_ratio/low_min": 0.05964052490890026, "clip_ratio/high_mean": 0.06107954680919647, "clip_ratio/high_max": 0.06107954680919647, "clip_ratio/region_mean": 0.12072007171809673, "reward_total_mean": 0.5795755982398987, "reward_meter_mean": 0.5983876585960388, "reward_meter_std": 0.34947502613067627, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8755115270614624, "reward_repeat_soft_std": 0.10106371343135834, "reward_judge_quality_mean": 0.24250000715255737, "reward_judge_quality_std": 0.11792854964733124, "reward_total_composite_mean": 0.5795755982398987, "reward_total_composite_std": 0.1329527199268341} {"timestamp_utc": "2026-04-13T00:16:15Z", "mode": "train", "global_step": 812, "epoch": 0.08156705173279759, "loss": 0.0486, "grad_norm": 15.196045875549316, "learning_rate": 7.542424242424244e-06, "num_tokens": 1487566.0, "completions/mean_length": 40.125, "completions/min_length": 33.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.125, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.7968858480453491, "rewards/meter/std": 0.31873252987861633, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7544649839401245, "rewards/repeat_soft/std": 0.2216007262468338, "rewards/judge_quality/mean": 0.2887499928474426, "rewards/judge_quality/std": 0.21931305527687073, "rewards/total_composite/mean": 0.6706701517105103, "rewards/total_composite/std": 0.17179952561855316, "reward": 0.6706701517105103, "reward_std": 0.17179951071739197, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10026001930236816, "sampling/sampling_logp_difference/max": 1.8747837543487549, "sampling/importance_sampling_ratio/min": 0.1533881425857544, "sampling/importance_sampling_ratio/mean": 1.0025943517684937, "sampling/importance_sampling_ratio/max": 1.8630106449127197, "entropy": 0.7375423200428486, "clip_ratio/low_mean": 0.02452153153717518, "clip_ratio/low_min": 0.02452153153717518, "clip_ratio/high_mean": 0.07814182667061687, "clip_ratio/high_max": 0.07814182667061687, "clip_ratio/region_mean": 0.10266335820779204, "reward_total_mean": 0.6706701517105103, "reward_meter_mean": 0.7968858480453491, "reward_meter_std": 0.31873252987861633, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7544649839401245, "reward_repeat_soft_std": 0.2216007262468338, "reward_judge_quality_mean": 0.2887499928474426, "reward_judge_quality_std": 0.21931305527687073, "reward_total_composite_mean": 0.6706701517105103, "reward_total_composite_std": 0.17179952561855316} {"timestamp_utc": "2026-04-13T00:16:31Z", "mode": "train", "global_step": 813, "epoch": 0.08166750376695128, "loss": -0.0943, "grad_norm": 2.3236231803894043, "learning_rate": 7.53939393939394e-06, "num_tokens": 1489136.0, "completions/mean_length": 165.25, "completions/min_length": 37.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 49.66666793823242, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.44218289852142334, "rewards/meter/std": 0.37565821409225464, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.2519763112068176, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9253338575363159, "rewards/repeat_soft/std": 0.0925251767039299, "rewards/judge_quality/mean": 0.23874999582767487, "rewards/judge_quality/std": 0.16287045180797577, "rewards/total_composite/mean": 0.4006338119506836, "rewards/total_composite/std": 0.2838401198387146, "reward": 0.4006338119506836, "reward_std": 0.2838401198387146, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1710057556629181, "sampling/sampling_logp_difference/max": 1.8904857635498047, "sampling/importance_sampling_ratio/min": 0.15099844336509705, "sampling/importance_sampling_ratio/mean": 1.0192328691482544, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7253069505095482, "clip_ratio/low_mean": 0.032590052112936974, "clip_ratio/low_min": 0.032590052112936974, "clip_ratio/high_mean": 0.07810442335903645, "clip_ratio/high_max": 0.07810442335903645, "clip_ratio/region_mean": 0.11069447547197342, "reward_total_mean": 0.4006338119506836, "reward_meter_mean": 0.44218289852142334, "reward_meter_std": 0.37565821409225464, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.2519763112068176, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9253338575363159, "reward_repeat_soft_std": 0.0925251767039299, "reward_judge_quality_mean": 0.23874999582767487, "reward_judge_quality_std": 0.16287045180797577, "reward_total_composite_mean": 0.4006338119506836, "reward_total_composite_std": 0.2838401198387146} {"timestamp_utc": "2026-04-13T00:16:45Z", "mode": "train", "global_step": 814, "epoch": 0.08176795580110498, "loss": -0.1433, "grad_norm": 3.186413526535034, "learning_rate": 7.536363636363637e-06, "num_tokens": 1490682.0, "completions/mean_length": 107.25, "completions/min_length": 31.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 49.42857360839844, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.7988342046737671, "rewards/meter/std": 0.3370206654071808, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8750088214874268, "rewards/repeat_soft/std": 0.11267176270484924, "rewards/judge_quality/mean": 0.26124998927116394, "rewards/judge_quality/std": 0.1600390523672104, "rewards/total_composite/mean": 0.6311397552490234, "rewards/total_composite/std": 0.27582842111587524, "reward": 0.6311397552490234, "reward_std": 0.27582842111587524, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1526871770620346, "sampling/sampling_logp_difference/max": 1.254561424255371, "sampling/importance_sampling_ratio/min": 0.28520089387893677, "sampling/importance_sampling_ratio/mean": 1.0252693891525269, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8987995684146881, "clip_ratio/low_mean": 0.013513513840734959, "clip_ratio/low_min": 0.013513513840734959, "clip_ratio/high_mean": 0.09632712835446, "clip_ratio/high_max": 0.09632712835446, "clip_ratio/region_mean": 0.10984064219519496, "reward_total_mean": 0.6311397552490234, "reward_meter_mean": 0.7988342046737671, "reward_meter_std": 0.3370206654071808, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8750088214874268, "reward_repeat_soft_std": 0.11267176270484924, "reward_judge_quality_mean": 0.26124998927116394, "reward_judge_quality_std": 0.1600390523672104, "reward_total_composite_mean": 0.6311397552490234, "reward_total_composite_std": 0.27582842111587524} {"timestamp_utc": "2026-04-13T00:16:52Z", "mode": "train", "global_step": 815, "epoch": 0.08186840783525866, "loss": 0.0381, "grad_norm": 16.25836181640625, "learning_rate": 7.533333333333334e-06, "num_tokens": 1492302.0, "completions/mean_length": 19.5, "completions/min_length": 18.0, "completions/max_length": 20.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.5, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 20.0, "rewards/meter/mean": 0.9979230165481567, "rewards/meter/std": 0.0028421790339052677, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9306918978691101, "rewards/repeat_soft/std": 0.05188051611185074, "rewards/judge_quality/mean": 0.9200000166893005, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.9681345820426941, "rewards/total_composite/std": 0.005265943706035614, "reward": 0.9681345820426941, "reward_std": 0.005265939049422741, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10545790940523148, "sampling/sampling_logp_difference/max": 1.9939556121826172, "sampling/importance_sampling_ratio/min": 0.13615578413009644, "sampling/importance_sampling_ratio/mean": 1.0026437044143677, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3609463609755039, "clip_ratio/low_mean": 0.025000000838190317, "clip_ratio/low_min": 0.025000000838190317, "clip_ratio/high_mean": 0.045394737273454666, "clip_ratio/high_max": 0.045394737273454666, "clip_ratio/region_mean": 0.07039473811164498, "reward_total_mean": 0.9681345820426941, "reward_meter_mean": 0.9979230165481567, "reward_meter_std": 0.0028421790339052677, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9306918978691101, "reward_repeat_soft_std": 0.05188051611185074, "reward_judge_quality_mean": 0.9200000166893005, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.9681345820426941, "reward_total_composite_std": 0.005265943706035614} {"timestamp_utc": "2026-04-13T00:16:58Z", "mode": "train", "global_step": 816, "epoch": 0.08196885986941235, "loss": 0.0095, "grad_norm": 12.960418701171875, "learning_rate": 7.530303030303031e-06, "num_tokens": 1493854.0, "completions/mean_length": 38.0, "completions/min_length": 32.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.9716724753379822, "rewards/meter/std": 0.03912109509110451, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7506043910980225, "rewards/repeat_soft/std": 0.2223624736070633, "rewards/judge_quality/mean": 0.39625000953674316, "rewards/judge_quality/std": 0.09085899591445923, "rewards/total_composite/mean": 0.7811880111694336, "rewards/total_composite/std": 0.03753958269953728, "reward": 0.7811880111694336, "reward_std": 0.03753958269953728, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12327927350997925, "sampling/sampling_logp_difference/max": 2.1532201766967773, "sampling/importance_sampling_ratio/min": 0.11610966175794601, "sampling/importance_sampling_ratio/mean": 0.9995249509811401, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6677455417811871, "clip_ratio/low_mean": 0.05263766832649708, "clip_ratio/low_min": 0.05263766832649708, "clip_ratio/high_mean": 0.08152174018323421, "clip_ratio/high_max": 0.08152174018323421, "clip_ratio/region_mean": 0.1341594085097313, "reward_total_mean": 0.7811880111694336, "reward_meter_mean": 0.9716724753379822, "reward_meter_std": 0.03912109509110451, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7506043910980225, "reward_repeat_soft_std": 0.2223624736070633, "reward_judge_quality_mean": 0.39625000953674316, "reward_judge_quality_std": 0.09085899591445923, "reward_total_composite_mean": 0.7811880111694336, "reward_total_composite_std": 0.03753958269953728} {"timestamp_utc": "2026-04-13T00:17:07Z", "mode": "train", "global_step": 817, "epoch": 0.08206931190356605, "loss": 0.001, "grad_norm": 8.830878257751465, "learning_rate": 7.5272727272727274e-06, "num_tokens": 1495869.0, "completions/mean_length": 62.875, "completions/min_length": 49.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.875, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.9904930591583252, "rewards/meter/std": 0.0034388024359941483, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.509174644947052, "rewards/repeat_soft/std": 0.09939853847026825, "rewards/judge_quality/mean": 0.23624999821186066, "rewards/judge_quality/std": 0.1246638298034668, "rewards/total_composite/mean": 0.6893893480300903, "rewards/total_composite/std": 0.032689016312360764, "reward": 0.6893893480300903, "reward_std": 0.03268900513648987, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09962885826826096, "sampling/sampling_logp_difference/max": 3.025970220565796, "sampling/importance_sampling_ratio/min": 0.048510730266571045, "sampling/importance_sampling_ratio/mean": 1.0136313438415527, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47757283598184586, "clip_ratio/low_mean": 0.03439100459218025, "clip_ratio/low_min": 0.03439100459218025, "clip_ratio/high_mean": 0.042190071661025286, "clip_ratio/high_max": 0.042190071661025286, "clip_ratio/region_mean": 0.07658107625320554, "reward_total_mean": 0.6893893480300903, "reward_meter_mean": 0.9904930591583252, "reward_meter_std": 0.0034388024359941483, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.509174644947052, "reward_repeat_soft_std": 0.09939853847026825, "reward_judge_quality_mean": 0.23624999821186066, "reward_judge_quality_std": 0.1246638298034668, "reward_total_composite_mean": 0.6893893480300903, "reward_total_composite_std": 0.032689016312360764} {"timestamp_utc": "2026-04-13T00:17:13Z", "mode": "train", "global_step": 818, "epoch": 0.08216976393771974, "loss": 0.0109, "grad_norm": 16.554201126098633, "learning_rate": 7.524242424242425e-06, "num_tokens": 1497419.0, "completions/mean_length": 35.75, "completions/min_length": 29.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.75, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.9060852527618408, "rewards/meter/std": 0.12977276742458344, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.693786084651947, "rewards/repeat_soft/std": 0.23353177309036255, "rewards/judge_quality/mean": 0.39250001311302185, "rewards/judge_quality/std": 0.25783440470695496, "rewards/total_composite/mean": 0.7448669672012329, "rewards/total_composite/std": 0.052911825478076935, "reward": 0.7448669672012329, "reward_std": 0.052911825478076935, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1270134449005127, "sampling/sampling_logp_difference/max": 1.6379122734069824, "sampling/importance_sampling_ratio/min": 0.19438543915748596, "sampling/importance_sampling_ratio/mean": 1.0083388090133667, "sampling/importance_sampling_ratio/max": 1.8715434074401855, "entropy": 0.6427725590765476, "clip_ratio/low_mean": 0.045685173477977514, "clip_ratio/low_min": 0.045685173477977514, "clip_ratio/high_mean": 0.08539118524640799, "clip_ratio/high_max": 0.08539118524640799, "clip_ratio/region_mean": 0.1310763587243855, "reward_total_mean": 0.7448669672012329, "reward_meter_mean": 0.9060852527618408, "reward_meter_std": 0.12977276742458344, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.693786084651947, "reward_repeat_soft_std": 0.23353177309036255, "reward_judge_quality_mean": 0.39250001311302185, "reward_judge_quality_std": 0.25783440470695496, "reward_total_composite_mean": 0.7448669672012329, "reward_total_composite_std": 0.052911825478076935} {"timestamp_utc": "2026-04-13T00:17:22Z", "mode": "train", "global_step": 819, "epoch": 0.08227021597187344, "loss": 0.081, "grad_norm": 9.464967727661133, "learning_rate": 7.521212121212121e-06, "num_tokens": 1499340.0, "completions/mean_length": 62.125, "completions/min_length": 47.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.125, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9785075783729553, "rewards/meter/std": 0.03234706073999405, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.5550739169120789, "rewards/repeat_soft/std": 0.16676342487335205, "rewards/judge_quality/mean": 0.24249999225139618, "rewards/judge_quality/std": 0.11792854964733124, "rewards/total_composite/mean": 0.7123357653617859, "rewards/total_composite/std": 0.0534050352871418, "reward": 0.7123357653617859, "reward_std": 0.0534050427377224, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09301181882619858, "sampling/sampling_logp_difference/max": 1.2385501861572266, "sampling/importance_sampling_ratio/min": 0.2898041009902954, "sampling/importance_sampling_ratio/mean": 1.014169692993164, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.588932704180479, "clip_ratio/low_mean": 0.04282791120931506, "clip_ratio/low_min": 0.04282791120931506, "clip_ratio/high_mean": 0.05584490764886141, "clip_ratio/high_max": 0.05584490764886141, "clip_ratio/region_mean": 0.09867281885817647, "reward_total_mean": 0.7123357653617859, "reward_meter_mean": 0.9785075783729553, "reward_meter_std": 0.03234706073999405, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.5550739169120789, "reward_repeat_soft_std": 0.16676342487335205, "reward_judge_quality_mean": 0.24249999225139618, "reward_judge_quality_std": 0.11792854964733124, "reward_total_composite_mean": 0.7123357653617859, "reward_total_composite_std": 0.0534050352871418} {"timestamp_utc": "2026-04-13T00:17:28Z", "mode": "train", "global_step": 820, "epoch": 0.08237066800602712, "loss": 0.017, "grad_norm": 15.16028118133545, "learning_rate": 7.518181818181819e-06, "num_tokens": 1501267.0, "completions/mean_length": 54.875, "completions/min_length": 46.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.875, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.875468909740448, "rewards/meter/std": 0.3016623258590698, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.6230003237724304, "rewards/repeat_soft/std": 0.1375703364610672, "rewards/judge_quality/mean": 0.1899999976158142, "rewards/judge_quality/std": 0.10184020549058914, "rewards/total_composite/mean": 0.5882388353347778, "rewards/total_composite/std": 0.24032209813594818, "reward": 0.5882388353347778, "reward_std": 0.24032209813594818, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09352630376815796, "sampling/sampling_logp_difference/max": 1.5539894104003906, "sampling/importance_sampling_ratio/min": 0.21140290796756744, "sampling/importance_sampling_ratio/mean": 1.0149915218353271, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.506966408342123, "clip_ratio/low_mean": 0.01785714365541935, "clip_ratio/low_min": 0.01785714365541935, "clip_ratio/high_mean": 0.07068923278711736, "clip_ratio/high_max": 0.07068923278711736, "clip_ratio/region_mean": 0.08854637644253671, "reward_total_mean": 0.5882388353347778, "reward_meter_mean": 0.875468909740448, "reward_meter_std": 0.3016623258590698, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.6230003237724304, "reward_repeat_soft_std": 0.1375703364610672, "reward_judge_quality_mean": 0.1899999976158142, "reward_judge_quality_std": 0.10184020549058914, "reward_total_composite_mean": 0.5882388353347778, "reward_total_composite_std": 0.24032209813594818} {"timestamp_utc": "2026-04-13T00:17:34Z", "mode": "train", "global_step": 821, "epoch": 0.08247112004018081, "loss": -0.0062, "grad_norm": 14.174189567565918, "learning_rate": 7.515151515151516e-06, "num_tokens": 1503112.0, "completions/mean_length": 47.625, "completions/min_length": 41.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.625, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.655773401260376, "rewards/meter/std": 0.3483875095844269, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.4913419485092163, "rewards/repeat_soft/std": 0.20081916451454163, "rewards/judge_quality/mean": 0.20875000953674316, "rewards/judge_quality/std": 0.0965752825140953, "rewards/total_composite/mean": 0.5193572044372559, "rewards/total_composite/std": 0.15974822640419006, "reward": 0.5193572044372559, "reward_std": 0.15974822640419006, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10723876953125, "sampling/sampling_logp_difference/max": 1.698553204536438, "sampling/importance_sampling_ratio/min": 0.1829480230808258, "sampling/importance_sampling_ratio/mean": 1.0145440101623535, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6240472011268139, "clip_ratio/low_mean": 0.039724511094391346, "clip_ratio/low_min": 0.039724511094391346, "clip_ratio/high_mean": 0.046455972362309694, "clip_ratio/high_max": 0.046455972362309694, "clip_ratio/region_mean": 0.08618048345670104, "reward_total_mean": 0.5193572044372559, "reward_meter_mean": 0.655773401260376, "reward_meter_std": 0.3483875095844269, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.4913419485092163, "reward_repeat_soft_std": 0.20081916451454163, "reward_judge_quality_mean": 0.20875000953674316, "reward_judge_quality_std": 0.0965752825140953, "reward_total_composite_mean": 0.5193572044372559, "reward_total_composite_std": 0.15974822640419006} {"timestamp_utc": "2026-04-13T00:17:42Z", "mode": "train", "global_step": 822, "epoch": 0.0825715720743345, "loss": -0.0411, "grad_norm": 14.882509231567383, "learning_rate": 7.512121212121213e-06, "num_tokens": 1504700.0, "completions/mean_length": 24.5, "completions/min_length": 20.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.5, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.8496226668357849, "rewards/meter/std": 0.23511943221092224, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8822308778762817, "rewards/repeat_soft/std": 0.0803644061088562, "rewards/judge_quality/mean": 0.3087499737739563, "rewards/judge_quality/std": 0.1141975075006485, "rewards/total_composite/mean": 0.7131783366203308, "rewards/total_composite/std": 0.13221338391304016, "reward": 0.7131783366203308, "reward_std": 0.13221336901187897, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15131300687789917, "sampling/sampling_logp_difference/max": 1.922037124633789, "sampling/importance_sampling_ratio/min": 0.14630861580371857, "sampling/importance_sampling_ratio/mean": 1.0294702053070068, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9560103416442871, "clip_ratio/low_mean": 0.05380434915423393, "clip_ratio/low_min": 0.05380434915423393, "clip_ratio/high_mean": 0.12099021347239614, "clip_ratio/high_max": 0.12099021347239614, "clip_ratio/region_mean": 0.17479456262663007, "reward_total_mean": 0.7131783366203308, "reward_meter_mean": 0.8496226668357849, "reward_meter_std": 0.23511943221092224, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8822308778762817, "reward_repeat_soft_std": 0.0803644061088562, "reward_judge_quality_mean": 0.3087499737739563, "reward_judge_quality_std": 0.1141975075006485, "reward_total_composite_mean": 0.7131783366203308, "reward_total_composite_std": 0.13221338391304016} {"timestamp_utc": "2026-04-13T00:17:52Z", "mode": "train", "global_step": 823, "epoch": 0.0826720241084882, "loss": 0.0405, "grad_norm": 8.141386985778809, "learning_rate": 7.509090909090909e-06, "num_tokens": 1507081.0, "completions/mean_length": 85.625, "completions/min_length": 71.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.625, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.8883504271507263, "rewards/meter/std": 0.20128075778484344, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8325358629226685, "rewards/repeat_soft/std": 0.13863785564899445, "rewards/judge_quality/mean": 0.2824999690055847, "rewards/judge_quality/std": 0.12578324973583221, "rewards/total_composite/mean": 0.6877613067626953, "rewards/total_composite/std": 0.08713164925575256, "reward": 0.6877613067626953, "reward_std": 0.08713164180517197, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11454110592603683, "sampling/sampling_logp_difference/max": 1.5117957592010498, "sampling/importance_sampling_ratio/min": 0.22051364183425903, "sampling/importance_sampling_ratio/mean": 1.0121146440505981, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6817474402487278, "clip_ratio/low_mean": 0.04015436535701156, "clip_ratio/low_min": 0.04015436535701156, "clip_ratio/high_mean": 0.0556839918717742, "clip_ratio/high_max": 0.0556839918717742, "clip_ratio/region_mean": 0.09583835722878575, "reward_total_mean": 0.6877613067626953, "reward_meter_mean": 0.8883504271507263, "reward_meter_std": 0.20128075778484344, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8325358629226685, "reward_repeat_soft_std": 0.13863785564899445, "reward_judge_quality_mean": 0.2824999690055847, "reward_judge_quality_std": 0.12578324973583221, "reward_total_composite_mean": 0.6877613067626953, "reward_total_composite_std": 0.08713164925575256} {"timestamp_utc": "2026-04-13T00:17:59Z", "mode": "train", "global_step": 824, "epoch": 0.08277247614264188, "loss": -0.0259, "grad_norm": 15.442203521728516, "learning_rate": 7.5060606060606065e-06, "num_tokens": 1508746.0, "completions/mean_length": 28.125, "completions/min_length": 23.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.125, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.8102967143058777, "rewards/meter/std": 0.22375255823135376, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8373357057571411, "rewards/repeat_soft/std": 0.09206025302410126, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.19334925711154938, "rewards/total_composite/mean": 0.7389920949935913, "rewards/total_composite/std": 0.12032730877399445, "reward": 0.7389920949935913, "reward_std": 0.12032730877399445, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.155991330742836, "sampling/sampling_logp_difference/max": 1.8317909240722656, "sampling/importance_sampling_ratio/min": 0.16012653708457947, "sampling/importance_sampling_ratio/mean": 0.972854733467102, "sampling/importance_sampling_ratio/max": 1.6852415800094604, "entropy": 0.5829095542430878, "clip_ratio/low_mean": 0.04956439416855574, "clip_ratio/low_min": 0.04956439416855574, "clip_ratio/high_mean": 0.07234618859365582, "clip_ratio/high_max": 0.07234618859365582, "clip_ratio/region_mean": 0.12191058276221156, "reward_total_mean": 0.7389920949935913, "reward_meter_mean": 0.8102967143058777, "reward_meter_std": 0.22375255823135376, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8373357057571411, "reward_repeat_soft_std": 0.09206025302410126, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.19334925711154938, "reward_total_composite_mean": 0.7389920949935913, "reward_total_composite_std": 0.12032730877399445} {"timestamp_utc": "2026-04-13T00:18:08Z", "mode": "train", "global_step": 825, "epoch": 0.08287292817679558, "loss": 0.0399, "grad_norm": 16.643545150756836, "learning_rate": 7.503030303030303e-06, "num_tokens": 1510473.0, "completions/mean_length": 34.875, "completions/min_length": 30.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.875, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.8256691694259644, "rewards/meter/std": 0.3352835774421692, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7480956315994263, "rewards/repeat_soft/std": 0.1840197741985321, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.14320313930511475, "rewards/total_composite/mean": 0.6931107044219971, "rewards/total_composite/std": 0.16561220586299896, "reward": 0.6931107044219971, "reward_std": 0.16561219096183777, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11857078224420547, "sampling/sampling_logp_difference/max": 4.026004791259766, "sampling/importance_sampling_ratio/min": 0.017845485359430313, "sampling/importance_sampling_ratio/mean": 1.022424578666687, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6722455509006977, "clip_ratio/low_mean": 0.0618484562728554, "clip_ratio/low_min": 0.0618484562728554, "clip_ratio/high_mean": 0.04869866417720914, "clip_ratio/high_max": 0.04869866417720914, "clip_ratio/region_mean": 0.11054712045006454, "reward_total_mean": 0.6931107044219971, "reward_meter_mean": 0.8256691694259644, "reward_meter_std": 0.3352835774421692, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7480956315994263, "reward_repeat_soft_std": 0.1840197741985321, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.14320313930511475, "reward_total_composite_mean": 0.6931107044219971, "reward_total_composite_std": 0.16561220586299896} {"timestamp_utc": "2026-04-13T00:18:14Z", "mode": "train", "global_step": 826, "epoch": 0.08297338021094927, "loss": -0.0387, "grad_norm": 8.790999412536621, "learning_rate": 7.500000000000001e-06, "num_tokens": 1512229.0, "completions/mean_length": 50.5, "completions/min_length": 40.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.5, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9800426363945007, "rewards/meter/std": 0.03276834264397621, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.17817415297031403, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.34563344717025757, "rewards/repeat_soft/std": 0.10055135935544968, "rewards/judge_quality/mean": 0.18125000596046448, "rewards/judge_quality/std": 0.07039429992437363, "rewards/total_composite/mean": 0.6549575328826904, "rewards/total_composite/std": 0.06109974905848503, "reward": 0.6549575328826904, "reward_std": 0.061099741607904434, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07667212188243866, "sampling/sampling_logp_difference/max": 0.9256272315979004, "sampling/importance_sampling_ratio/min": 0.39628279209136963, "sampling/importance_sampling_ratio/mean": 1.0022121667861938, "sampling/importance_sampling_ratio/max": 1.7144050598144531, "entropy": 0.5151231363415718, "clip_ratio/low_mean": 0.028157596476376057, "clip_ratio/low_min": 0.028157596476376057, "clip_ratio/high_mean": 0.04405159433372319, "clip_ratio/high_max": 0.04405159433372319, "clip_ratio/region_mean": 0.07220919081009924, "reward_total_mean": 0.6549575328826904, "reward_meter_mean": 0.9800426363945007, "reward_meter_std": 0.03276834264397621, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.17817415297031403, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.34563344717025757, "reward_repeat_soft_std": 0.10055135935544968, "reward_judge_quality_mean": 0.18125000596046448, "reward_judge_quality_std": 0.07039429992437363, "reward_total_composite_mean": 0.6549575328826904, "reward_total_composite_std": 0.06109974905848503} {"timestamp_utc": "2026-04-13T00:18:21Z", "mode": "train", "global_step": 827, "epoch": 0.08307383224510297, "loss": 0.0351, "grad_norm": 17.53249168395996, "learning_rate": 7.496969696969698e-06, "num_tokens": 1513710.0, "completions/mean_length": 35.125, "completions/min_length": 28.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.125, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.35525834560394287, "rewards/meter/std": 0.26282691955566406, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8827140927314758, "rewards/repeat_soft/std": 0.14737999439239502, "rewards/judge_quality/mean": 0.2462499886751175, "rewards/judge_quality/std": 0.08348438143730164, "rewards/total_composite/mean": 0.4720126688480377, "rewards/total_composite/std": 0.12447939813137054, "reward": 0.4720126688480377, "reward_std": 0.12447939068078995, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1870914101600647, "sampling/sampling_logp_difference/max": 5.577481269836426, "sampling/importance_sampling_ratio/min": 0.003782079555094242, "sampling/importance_sampling_ratio/mean": 0.9943389296531677, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7316368632018566, "clip_ratio/low_mean": 0.13005812279880047, "clip_ratio/low_min": 0.13005812279880047, "clip_ratio/high_mean": 0.05396299157291651, "clip_ratio/high_max": 0.05396299157291651, "clip_ratio/region_mean": 0.18402111437171698, "reward_total_mean": 0.4720126688480377, "reward_meter_mean": 0.35525834560394287, "reward_meter_std": 0.26282691955566406, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8827140927314758, "reward_repeat_soft_std": 0.14737999439239502, "reward_judge_quality_mean": 0.2462499886751175, "reward_judge_quality_std": 0.08348438143730164, "reward_total_composite_mean": 0.4720126688480377, "reward_total_composite_std": 0.12447939813137054} {"timestamp_utc": "2026-04-13T00:18:28Z", "mode": "train", "global_step": 828, "epoch": 0.08317428427925666, "loss": 0.0613, "grad_norm": 13.418478012084961, "learning_rate": 7.493939393939395e-06, "num_tokens": 1515402.0, "completions/mean_length": 53.5, "completions/min_length": 30.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.5, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.6146317720413208, "rewards/meter/std": 0.3963913321495056, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9311263561248779, "rewards/repeat_soft/std": 0.09322485327720642, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.3341701030731201, "rewards/total_composite/mean": 0.6790719032287598, "rewards/total_composite/std": 0.2154712975025177, "reward": 0.6790719032287598, "reward_std": 0.2154712826013565, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13674475252628326, "sampling/sampling_logp_difference/max": 1.3601908683776855, "sampling/importance_sampling_ratio/min": 0.25661179423332214, "sampling/importance_sampling_ratio/mean": 0.9992030262947083, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.755468375980854, "clip_ratio/low_mean": 0.033541055396199226, "clip_ratio/low_min": 0.033541055396199226, "clip_ratio/high_mean": 0.08326868619769812, "clip_ratio/high_max": 0.08326868619769812, "clip_ratio/region_mean": 0.11680974159389734, "reward_total_mean": 0.6790719032287598, "reward_meter_mean": 0.6146317720413208, "reward_meter_std": 0.3963913321495056, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9311263561248779, "reward_repeat_soft_std": 0.09322485327720642, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.3341701030731201, "reward_total_composite_mean": 0.6790719032287598, "reward_total_composite_std": 0.2154712975025177} {"timestamp_utc": "2026-04-13T00:18:35Z", "mode": "train", "global_step": 829, "epoch": 0.08327473631341034, "loss": 0.0162, "grad_norm": 20.183799743652344, "learning_rate": 7.490909090909092e-06, "num_tokens": 1516773.0, "completions/mean_length": 23.375, "completions/min_length": 18.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.375, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.7565333247184753, "rewards/meter/std": 0.18880954384803772, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9326152801513672, "rewards/repeat_soft/std": 0.1299283355474472, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.7067015171051025, "rewards/total_composite/std": 0.09154196828603745, "reward": 0.7067015171051025, "reward_std": 0.09154196083545685, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14223571121692657, "sampling/sampling_logp_difference/max": 2.2191245555877686, "sampling/importance_sampling_ratio/min": 0.10870423167943954, "sampling/importance_sampling_ratio/mean": 1.0132664442062378, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7161458432674408, "clip_ratio/low_mean": 0.032866379246115685, "clip_ratio/low_min": 0.032866379246115685, "clip_ratio/high_mean": 0.09581044130027294, "clip_ratio/high_max": 0.09581044130027294, "clip_ratio/region_mean": 0.12867682054638863, "reward_total_mean": 0.7067015171051025, "reward_meter_mean": 0.7565333247184753, "reward_meter_std": 0.18880954384803772, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9326152801513672, "reward_repeat_soft_std": 0.1299283355474472, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.7067015171051025, "reward_total_composite_std": 0.09154196828603745} {"timestamp_utc": "2026-04-13T00:18:44Z", "mode": "train", "global_step": 830, "epoch": 0.08337518834756404, "loss": -0.0193, "grad_norm": 7.151583671569824, "learning_rate": 7.487878787878788e-06, "num_tokens": 1518980.0, "completions/mean_length": 74.875, "completions/min_length": 62.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.875, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9739394187927246, "rewards/meter/std": 0.01722760498523712, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.4482308030128479, "rewards/repeat_soft/std": 0.1938052922487259, "rewards/judge_quality/mean": 0.14375001192092896, "rewards/judge_quality/std": 0.0176776722073555, "rewards/total_composite/mean": 0.6462208032608032, "rewards/total_composite/std": 0.022095201537013054, "reward": 0.6462208032608032, "reward_std": 0.022095195949077606, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07253661006689072, "sampling/sampling_logp_difference/max": 1.8998427391052246, "sampling/importance_sampling_ratio/min": 0.14959214627742767, "sampling/importance_sampling_ratio/mean": 1.0028969049453735, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45213751308619976, "clip_ratio/low_mean": 0.02942167455330491, "clip_ratio/low_min": 0.02942167455330491, "clip_ratio/high_mean": 0.050906818360090256, "clip_ratio/high_max": 0.050906818360090256, "clip_ratio/region_mean": 0.08032849291339517, "reward_total_mean": 0.6462208032608032, "reward_meter_mean": 0.9739394187927246, "reward_meter_std": 0.01722760498523712, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.4482308030128479, "reward_repeat_soft_std": 0.1938052922487259, "reward_judge_quality_mean": 0.14375001192092896, "reward_judge_quality_std": 0.0176776722073555, "reward_total_composite_mean": 0.6462208032608032, "reward_total_composite_std": 0.022095201537013054} {"timestamp_utc": "2026-04-13T00:18:53Z", "mode": "train", "global_step": 831, "epoch": 0.08347564038171773, "loss": -0.1183, "grad_norm": 8.570809364318848, "learning_rate": 7.484848484848486e-06, "num_tokens": 1520831.0, "completions/mean_length": 51.375, "completions/min_length": 35.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.375, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.992888331413269, "rewards/meter/std": 0.0010251014027744532, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.5650733709335327, "rewards/repeat_soft/std": 0.11576336622238159, "rewards/judge_quality/mean": 0.22499999403953552, "rewards/judge_quality/std": 0.04629100486636162, "rewards/total_composite/mean": 0.7083070874214172, "rewards/total_composite/std": 0.04043668136000633, "reward": 0.7083070874214172, "reward_std": 0.04043668508529663, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0996473878622055, "sampling/sampling_logp_difference/max": 1.5513792037963867, "sampling/importance_sampling_ratio/min": 0.21195544302463531, "sampling/importance_sampling_ratio/mean": 1.0057207345962524, "sampling/importance_sampling_ratio/max": 1.8468518257141113, "entropy": 0.6002024114131927, "clip_ratio/low_mean": 0.024614846101030707, "clip_ratio/low_min": 0.024614846101030707, "clip_ratio/high_mean": 0.07775143627077341, "clip_ratio/high_max": 0.07775143627077341, "clip_ratio/region_mean": 0.10236628237180412, "reward_total_mean": 0.7083070874214172, "reward_meter_mean": 0.992888331413269, "reward_meter_std": 0.0010251014027744532, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.5650733709335327, "reward_repeat_soft_std": 0.11576336622238159, "reward_judge_quality_mean": 0.22499999403953552, "reward_judge_quality_std": 0.04629100486636162, "reward_total_composite_mean": 0.7083070874214172, "reward_total_composite_std": 0.04043668136000633} {"timestamp_utc": "2026-04-13T00:19:00Z", "mode": "train", "global_step": 832, "epoch": 0.08357609241587143, "loss": 0.0972, "grad_norm": 27.777599334716797, "learning_rate": 7.481818181818182e-06, "num_tokens": 1522201.0, "completions/mean_length": 21.25, "completions/min_length": 18.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.25, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.5177667737007141, "rewards/meter/std": 0.37568557262420654, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9140229225158691, "rewards/repeat_soft/std": 0.06058330833911896, "rewards/judge_quality/mean": 0.26124998927116394, "rewards/judge_quality/std": 0.28767234086990356, "rewards/total_composite/mean": 0.5527723431587219, "rewards/total_composite/std": 0.20811593532562256, "reward": 0.5527723431587219, "reward_std": 0.20811593532562256, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22255048155784607, "sampling/sampling_logp_difference/max": 1.7417993545532227, "sampling/importance_sampling_ratio/min": 0.17520485818386078, "sampling/importance_sampling_ratio/mean": 1.0227082967758179, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3542979583144188, "clip_ratio/low_mean": 0.06009615492075682, "clip_ratio/low_min": 0.06009615492075682, "clip_ratio/high_mean": 0.11462843045592308, "clip_ratio/high_max": 0.11462843045592308, "clip_ratio/region_mean": 0.1747245853766799, "reward_total_mean": 0.5527723431587219, "reward_meter_mean": 0.5177667737007141, "reward_meter_std": 0.37568557262420654, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9140229225158691, "reward_repeat_soft_std": 0.06058330833911896, "reward_judge_quality_mean": 0.26124998927116394, "reward_judge_quality_std": 0.28767234086990356, "reward_total_composite_mean": 0.5527723431587219, "reward_total_composite_std": 0.20811593532562256} {"timestamp_utc": "2026-04-13T00:19:08Z", "mode": "train", "global_step": 833, "epoch": 0.08367654445002512, "loss": 0.0054, "grad_norm": 12.100703239440918, "learning_rate": 7.47878787878788e-06, "num_tokens": 1523943.0, "completions/mean_length": 51.75, "completions/min_length": 44.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.75, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9890915751457214, "rewards/meter/std": 0.002967091277241707, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7850879430770874, "rewards/repeat_soft/std": 0.08825700730085373, "rewards/judge_quality/mean": 0.2462500035762787, "rewards/judge_quality/std": 0.08348438143730164, "rewards/total_composite/mean": 0.709975004196167, "rewards/total_composite/std": 0.03315194323658943, "reward": 0.709975004196167, "reward_std": 0.03315194323658943, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11846042424440384, "sampling/sampling_logp_difference/max": 2.7437148094177246, "sampling/importance_sampling_ratio/min": 0.06433092057704926, "sampling/importance_sampling_ratio/mean": 1.018845796585083, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5846834108233452, "clip_ratio/low_mean": 0.053304925095289946, "clip_ratio/low_min": 0.053304925095289946, "clip_ratio/high_mean": 0.036352901719510555, "clip_ratio/high_max": 0.036352901719510555, "clip_ratio/region_mean": 0.0896578268148005, "reward_total_mean": 0.709975004196167, "reward_meter_mean": 0.9890915751457214, "reward_meter_std": 0.002967091277241707, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7850879430770874, "reward_repeat_soft_std": 0.08825700730085373, "reward_judge_quality_mean": 0.2462500035762787, "reward_judge_quality_std": 0.08348438143730164, "reward_total_composite_mean": 0.709975004196167, "reward_total_composite_std": 0.03315194323658943} {"timestamp_utc": "2026-04-13T00:19:15Z", "mode": "train", "global_step": 834, "epoch": 0.0837769964841788, "loss": -0.0287, "grad_norm": 8.129668235778809, "learning_rate": 7.4757575757575765e-06, "num_tokens": 1525923.0, "completions/mean_length": 71.5, "completions/min_length": 43.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.5, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.8951735496520996, "rewards/meter/std": 0.26843929290771484, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8119722604751587, "rewards/repeat_soft/std": 0.10040795058012009, "rewards/judge_quality/mean": 0.2150000035762787, "rewards/judge_quality/std": 0.10113639384508133, "rewards/total_composite/mean": 0.6610252857208252, "rewards/total_composite/std": 0.1210373044013977, "reward": 0.6610252857208252, "reward_std": 0.12103728204965591, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12606795132160187, "sampling/sampling_logp_difference/max": 1.430419683456421, "sampling/importance_sampling_ratio/min": 0.23920850455760956, "sampling/importance_sampling_ratio/mean": 1.0138356685638428, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7952045798301697, "clip_ratio/low_mean": 0.026595745235681534, "clip_ratio/low_min": 0.026595745235681534, "clip_ratio/high_mean": 0.1113510993309319, "clip_ratio/high_max": 0.1113510993309319, "clip_ratio/region_mean": 0.13794684456661344, "reward_total_mean": 0.6610252857208252, "reward_meter_mean": 0.8951735496520996, "reward_meter_std": 0.26843929290771484, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8119722604751587, "reward_repeat_soft_std": 0.10040795058012009, "reward_judge_quality_mean": 0.2150000035762787, "reward_judge_quality_std": 0.10113639384508133, "reward_total_composite_mean": 0.6610252857208252, "reward_total_composite_std": 0.1210373044013977} {"timestamp_utc": "2026-04-13T00:19:22Z", "mode": "train", "global_step": 835, "epoch": 0.0838774485183325, "loss": 0.0408, "grad_norm": 9.312095642089844, "learning_rate": 7.472727272727274e-06, "num_tokens": 1527807.0, "completions/mean_length": 67.5, "completions/min_length": 57.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.5, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.8704489469528198, "rewards/meter/std": 0.2243984490633011, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7693066596984863, "rewards/repeat_soft/std": 0.198989599943161, "rewards/judge_quality/mean": 0.2462499886751175, "rewards/judge_quality/std": 0.08348438143730164, "rewards/total_composite/mean": 0.6625076532363892, "rewards/total_composite/std": 0.1191767081618309, "reward": 0.6625076532363892, "reward_std": 0.1191767007112503, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1261705458164215, "sampling/sampling_logp_difference/max": 2.633249282836914, "sampling/importance_sampling_ratio/min": 0.07184464484453201, "sampling/importance_sampling_ratio/mean": 1.0035845041275024, "sampling/importance_sampling_ratio/max": 1.969367504119873, "entropy": 0.8021209985017776, "clip_ratio/low_mean": 0.0229752897284925, "clip_ratio/low_min": 0.0229752897284925, "clip_ratio/high_mean": 0.11141996830701828, "clip_ratio/high_max": 0.11141996830701828, "clip_ratio/region_mean": 0.13439525803551078, "reward_total_mean": 0.6625076532363892, "reward_meter_mean": 0.8704489469528198, "reward_meter_std": 0.2243984490633011, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7693066596984863, "reward_repeat_soft_std": 0.198989599943161, "reward_judge_quality_mean": 0.2462499886751175, "reward_judge_quality_std": 0.08348438143730164, "reward_total_composite_mean": 0.6625076532363892, "reward_total_composite_std": 0.1191767081618309} {"timestamp_utc": "2026-04-13T00:19:33Z", "mode": "train", "global_step": 836, "epoch": 0.08397790055248619, "loss": 0.0657, "grad_norm": 22.47435188293457, "learning_rate": 7.46969696969697e-06, "num_tokens": 1529242.0, "completions/mean_length": 31.375, "completions/min_length": 28.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.375, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.594031572341919, "rewards/meter/std": 0.35937270522117615, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.984477162361145, "rewards/repeat_soft/std": 0.026591889560222626, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.1940544992685318, "rewards/total_composite/mean": 0.6552619338035583, "rewards/total_composite/std": 0.18595010042190552, "reward": 0.6552619338035583, "reward_std": 0.18595010042190552, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.150856152176857, "sampling/sampling_logp_difference/max": 1.8114943504333496, "sampling/importance_sampling_ratio/min": 0.1634097695350647, "sampling/importance_sampling_ratio/mean": 1.0069156885147095, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8679464384913445, "clip_ratio/low_mean": 0.04464499559253454, "clip_ratio/low_min": 0.04464499559253454, "clip_ratio/high_mean": 0.09391910303384066, "clip_ratio/high_max": 0.09391910303384066, "clip_ratio/region_mean": 0.1385640986263752, "reward_total_mean": 0.6552619338035583, "reward_meter_mean": 0.594031572341919, "reward_meter_std": 0.35937270522117615, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.984477162361145, "reward_repeat_soft_std": 0.026591889560222626, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.1940544992685318, "reward_total_composite_mean": 0.6552619338035583, "reward_total_composite_std": 0.18595010042190552} {"timestamp_utc": "2026-04-13T00:19:43Z", "mode": "train", "global_step": 837, "epoch": 0.08407835258663988, "loss": -0.0332, "grad_norm": 14.945502281188965, "learning_rate": 7.4666666666666675e-06, "num_tokens": 1530724.0, "completions/mean_length": 20.25, "completions/min_length": 15.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.25, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.8730649948120117, "rewards/meter/std": 0.18960855901241302, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8487696647644043, "rewards/repeat_soft/std": 0.1468421220779419, "rewards/judge_quality/mean": 0.2800000011920929, "rewards/judge_quality/std": 0.13125765323638916, "rewards/total_composite/mean": 0.6202676296234131, "rewards/total_composite/std": 0.2688041031360626, "reward": 0.6202676296234131, "reward_std": 0.2688041031360626, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16857045888900757, "sampling/sampling_logp_difference/max": 2.2842283248901367, "sampling/importance_sampling_ratio/min": 0.10185262560844421, "sampling/importance_sampling_ratio/mean": 0.9958630204200745, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9925682172179222, "clip_ratio/low_mean": 0.020833333488553762, "clip_ratio/low_min": 0.020833333488553762, "clip_ratio/high_mean": 0.08686007373034954, "clip_ratio/high_max": 0.08686007373034954, "clip_ratio/region_mean": 0.1076934072189033, "reward_total_mean": 0.6202676296234131, "reward_meter_mean": 0.8730649948120117, "reward_meter_std": 0.18960855901241302, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8487696647644043, "reward_repeat_soft_std": 0.1468421220779419, "reward_judge_quality_mean": 0.2800000011920929, "reward_judge_quality_std": 0.13125765323638916, "reward_total_composite_mean": 0.6202676296234131, "reward_total_composite_std": 0.2688041031360626} {"timestamp_utc": "2026-04-13T00:19:54Z", "mode": "train", "global_step": 838, "epoch": 0.08417880462079357, "loss": 0.0503, "grad_norm": 22.646696090698242, "learning_rate": 7.463636363636364e-06, "num_tokens": 1532120.0, "completions/mean_length": 22.5, "completions/min_length": 16.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.5, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9517356157302856, "rewards/meter/std": 0.10177984833717346, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8541568517684937, "rewards/repeat_soft/std": 0.12708351016044617, "rewards/judge_quality/mean": 0.36250001192092896, "rewards/judge_quality/std": 0.12291111797094345, "rewards/total_composite/mean": 0.7724466919898987, "rewards/total_composite/std": 0.08618444204330444, "reward": 0.7724466919898987, "reward_std": 0.08618444204330444, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1600654423236847, "sampling/sampling_logp_difference/max": 1.2735111713409424, "sampling/importance_sampling_ratio/min": 0.27984729409217834, "sampling/importance_sampling_ratio/mean": 1.0089377164840698, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8764792904257774, "clip_ratio/low_mean": 0.04832175932824612, "clip_ratio/low_min": 0.04832175932824612, "clip_ratio/high_mean": 0.08318749954923987, "clip_ratio/high_max": 0.08318749954923987, "clip_ratio/region_mean": 0.131509258877486, "reward_total_mean": 0.7724466919898987, "reward_meter_mean": 0.9517356157302856, "reward_meter_std": 0.10177984833717346, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8541568517684937, "reward_repeat_soft_std": 0.12708351016044617, "reward_judge_quality_mean": 0.36250001192092896, "reward_judge_quality_std": 0.12291111797094345, "reward_total_composite_mean": 0.7724466919898987, "reward_total_composite_std": 0.08618444204330444} {"timestamp_utc": "2026-04-13T00:20:04Z", "mode": "train", "global_step": 839, "epoch": 0.08427925665494726, "loss": 0.1076, "grad_norm": 12.030730247497559, "learning_rate": 7.460606060606061e-06, "num_tokens": 1534025.0, "completions/mean_length": 56.125, "completions/min_length": 41.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.125, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.6572591066360474, "rewards/meter/std": 0.352651447057724, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7561246156692505, "rewards/repeat_soft/std": 0.18780268728733063, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.5843790769577026, "rewards/total_composite/std": 0.16203273832798004, "reward": 0.5843790769577026, "reward_std": 0.16203273832798004, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1298743486404419, "sampling/sampling_logp_difference/max": 1.6454172134399414, "sampling/importance_sampling_ratio/min": 0.19293203949928284, "sampling/importance_sampling_ratio/mean": 1.015926480293274, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7054343745112419, "clip_ratio/low_mean": 0.05939517542719841, "clip_ratio/low_min": 0.05939517542719841, "clip_ratio/high_mean": 0.07252840977162123, "clip_ratio/high_max": 0.07252840977162123, "clip_ratio/region_mean": 0.13192358519881964, "reward_total_mean": 0.5843790769577026, "reward_meter_mean": 0.6572591066360474, "reward_meter_std": 0.352651447057724, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7561246156692505, "reward_repeat_soft_std": 0.18780268728733063, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.5843790769577026, "reward_total_composite_std": 0.16203273832798004} {"timestamp_utc": "2026-04-13T00:20:12Z", "mode": "train", "global_step": 840, "epoch": 0.08437970868910095, "loss": 0.2313, "grad_norm": 21.98044776916504, "learning_rate": 7.4575757575757575e-06, "num_tokens": 1535467.0, "completions/mean_length": 27.25, "completions/min_length": 11.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.25, "completions/min_terminated_length": 11.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.23164479434490204, "rewards/meter/std": 0.29977741837501526, "rewards/count_adherence/mean": 0.5, "rewards/count_adherence/std": 0.5345224738121033, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9631153345108032, "rewards/repeat_soft/std": 0.015462235547602177, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.391426682472229, "rewards/total_composite/std": 0.17140507698059082, "reward": 0.391426682472229, "reward_std": 0.17140507698059082, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20764656364917755, "sampling/sampling_logp_difference/max": 9.926230430603027, "sampling/importance_sampling_ratio/min": 4.8875692300498486e-05, "sampling/importance_sampling_ratio/mean": 1.0136806964874268, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7843061462044716, "clip_ratio/low_mean": 0.09338917955756187, "clip_ratio/low_min": 0.09338917955756187, "clip_ratio/high_mean": 0.06666666828095913, "clip_ratio/high_max": 0.06666666828095913, "clip_ratio/region_mean": 0.160055847838521, "reward_total_mean": 0.391426682472229, "reward_meter_mean": 0.23164479434490204, "reward_meter_std": 0.29977741837501526, "reward_count_adherence_mean": 0.5, "reward_count_adherence_std": 0.5345224738121033, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9631153345108032, "reward_repeat_soft_std": 0.015462235547602177, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.391426682472229, "reward_total_composite_std": 0.17140507698059082} {"timestamp_utc": "2026-04-13T00:20:19Z", "mode": "train", "global_step": 841, "epoch": 0.08448016072325465, "loss": 0.1062, "grad_norm": 10.94680404663086, "learning_rate": 7.454545454545456e-06, "num_tokens": 1537278.0, "completions/mean_length": 48.375, "completions/min_length": 35.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.375, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.505953848361969, "rewards/meter/std": 0.40242037177085876, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8316765427589417, "rewards/repeat_soft/std": 0.1457357555627823, "rewards/judge_quality/mean": 0.36000001430511475, "rewards/judge_quality/std": 0.09165150672197342, "rewards/total_composite/mean": 0.5563468933105469, "rewards/total_composite/std": 0.17831359803676605, "reward": 0.5563468933105469, "reward_std": 0.17831359803676605, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1264338195323944, "sampling/sampling_logp_difference/max": 2.188642740249634, "sampling/importance_sampling_ratio/min": 0.11206875741481781, "sampling/importance_sampling_ratio/mean": 0.9978600144386292, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5854948535561562, "clip_ratio/low_mean": 0.07228206936269999, "clip_ratio/low_min": 0.07228206936269999, "clip_ratio/high_mean": 0.05497217085212469, "clip_ratio/high_max": 0.05497217085212469, "clip_ratio/region_mean": 0.12725424021482468, "reward_total_mean": 0.5563468933105469, "reward_meter_mean": 0.505953848361969, "reward_meter_std": 0.40242037177085876, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8316765427589417, "reward_repeat_soft_std": 0.1457357555627823, "reward_judge_quality_mean": 0.36000001430511475, "reward_judge_quality_std": 0.09165150672197342, "reward_total_composite_mean": 0.5563468933105469, "reward_total_composite_std": 0.17831359803676605} {"timestamp_utc": "2026-04-13T00:20:31Z", "mode": "train", "global_step": 842, "epoch": 0.08458061275740834, "loss": -0.0615, "grad_norm": 11.976949691772461, "learning_rate": 7.451515151515152e-06, "num_tokens": 1539354.0, "completions/mean_length": 133.5, "completions/min_length": 58.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 79.42857360839844, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.7871052622795105, "rewards/meter/std": 0.2334442138671875, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.28171807527542114, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8461621999740601, "rewards/repeat_soft/std": 0.15790514647960663, "rewards/judge_quality/mean": 0.4625000059604645, "rewards/judge_quality/std": 0.3561199903488159, "rewards/total_composite/mean": 0.7025635838508606, "rewards/total_composite/std": 0.21455839276313782, "reward": 0.7025635838508606, "reward_std": 0.21455836296081543, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1475580334663391, "sampling/sampling_logp_difference/max": 2.462618589401245, "sampling/importance_sampling_ratio/min": 0.08521152287721634, "sampling/importance_sampling_ratio/mean": 1.0085952281951904, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6350515596568584, "clip_ratio/low_mean": 0.052589246071875095, "clip_ratio/low_min": 0.052589246071875095, "clip_ratio/high_mean": 0.06737250462174416, "clip_ratio/high_max": 0.06737250462174416, "clip_ratio/region_mean": 0.11996175069361925, "reward_total_mean": 0.7025635838508606, "reward_meter_mean": 0.7871052622795105, "reward_meter_std": 0.2334442138671875, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.28171807527542114, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8461621999740601, "reward_repeat_soft_std": 0.15790514647960663, "reward_judge_quality_mean": 0.4625000059604645, "reward_judge_quality_std": 0.3561199903488159, "reward_total_composite_mean": 0.7025635838508606, "reward_total_composite_std": 0.21455839276313782} {"timestamp_utc": "2026-04-13T00:20:38Z", "mode": "train", "global_step": 843, "epoch": 0.08468106479156202, "loss": -0.007, "grad_norm": 10.413354873657227, "learning_rate": 7.448484848484849e-06, "num_tokens": 1541417.0, "completions/mean_length": 65.875, "completions/min_length": 55.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.875, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.8716582655906677, "rewards/meter/std": 0.26807862520217896, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8207269906997681, "rewards/repeat_soft/std": 0.08377651125192642, "rewards/judge_quality/mean": 0.39750000834465027, "rewards/judge_quality/std": 0.2454005628824234, "rewards/total_composite/mean": 0.7060688734054565, "rewards/total_composite/std": 0.1678328961133957, "reward": 0.7060688734054565, "reward_std": 0.1678328812122345, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12693195044994354, "sampling/sampling_logp_difference/max": 2.064110279083252, "sampling/importance_sampling_ratio/min": 0.17464682459831238, "sampling/importance_sampling_ratio/mean": 0.9842132925987244, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7327460870146751, "clip_ratio/low_mean": 0.01657197019085288, "clip_ratio/low_min": 0.01657197019085288, "clip_ratio/high_mean": 0.12460104934871197, "clip_ratio/high_max": 0.12460104934871197, "clip_ratio/region_mean": 0.14117301953956485, "reward_total_mean": 0.7060688734054565, "reward_meter_mean": 0.8716582655906677, "reward_meter_std": 0.26807862520217896, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8207269906997681, "reward_repeat_soft_std": 0.08377651125192642, "reward_judge_quality_mean": 0.39750000834465027, "reward_judge_quality_std": 0.2454005628824234, "reward_total_composite_mean": 0.7060688734054565, "reward_total_composite_std": 0.1678328961133957} {"timestamp_utc": "2026-04-13T00:20:45Z", "mode": "train", "global_step": 844, "epoch": 0.08478151682571572, "loss": 0.038, "grad_norm": 18.97239112854004, "learning_rate": 7.445454545454546e-06, "num_tokens": 1543242.0, "completions/mean_length": 54.125, "completions/min_length": 49.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.125, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.5768202543258667, "rewards/meter/std": 0.2918715476989746, "rewards/count_adherence/mean": 0.8500000238418579, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9396011829376221, "rewards/repeat_soft/std": 0.043566592037677765, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.6070291996002197, "rewards/total_composite/std": 0.12761364877223969, "reward": 0.6070291996002197, "reward_std": 0.12761364877223969, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18245775997638702, "sampling/sampling_logp_difference/max": 2.482409954071045, "sampling/importance_sampling_ratio/min": 0.08354165405035019, "sampling/importance_sampling_ratio/mean": 0.9850602746009827, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8435636758804321, "clip_ratio/low_mean": 0.06268053222447634, "clip_ratio/low_min": 0.06268053222447634, "clip_ratio/high_mean": 0.08140208199620247, "clip_ratio/high_max": 0.08140208199620247, "clip_ratio/region_mean": 0.1440826142206788, "reward_total_mean": 0.6070291996002197, "reward_meter_mean": 0.5768202543258667, "reward_meter_std": 0.2918715476989746, "reward_count_adherence_mean": 0.8500000238418579, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9396011829376221, "reward_repeat_soft_std": 0.043566592037677765, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.6070291996002197, "reward_total_composite_std": 0.12761364877223969} {"timestamp_utc": "2026-04-13T00:20:54Z", "mode": "train", "global_step": 845, "epoch": 0.08488196885986941, "loss": -0.0306, "grad_norm": 17.205509185791016, "learning_rate": 7.442424242424243e-06, "num_tokens": 1544987.0, "completions/mean_length": 58.125, "completions/min_length": 48.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.125, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.7751617431640625, "rewards/meter/std": 0.4008215069770813, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7545756101608276, "rewards/repeat_soft/std": 0.1822185516357422, "rewards/judge_quality/mean": 0.1899999976158142, "rewards/judge_quality/std": 0.10184020549058914, "rewards/total_composite/mean": 0.6312803626060486, "rewards/total_composite/std": 0.1582270711660385, "reward": 0.6312803626060486, "reward_std": 0.15822705626487732, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09847192466259003, "sampling/sampling_logp_difference/max": 1.9938256740570068, "sampling/importance_sampling_ratio/min": 0.13617347180843353, "sampling/importance_sampling_ratio/mean": 1.0049998760223389, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6286648958921432, "clip_ratio/low_mean": 0.020294540096074343, "clip_ratio/low_min": 0.020294540096074343, "clip_ratio/high_mean": 0.06267818342894316, "clip_ratio/high_max": 0.06267818342894316, "clip_ratio/region_mean": 0.0829727235250175, "reward_total_mean": 0.6312803626060486, "reward_meter_mean": 0.7751617431640625, "reward_meter_std": 0.4008215069770813, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7545756101608276, "reward_repeat_soft_std": 0.1822185516357422, "reward_judge_quality_mean": 0.1899999976158142, "reward_judge_quality_std": 0.10184020549058914, "reward_total_composite_mean": 0.6312803626060486, "reward_total_composite_std": 0.1582270711660385} {"timestamp_utc": "2026-04-13T00:21:01Z", "mode": "train", "global_step": 846, "epoch": 0.08498242089402311, "loss": 0.0453, "grad_norm": 16.241798400878906, "learning_rate": 7.439393939393939e-06, "num_tokens": 1546618.0, "completions/mean_length": 35.875, "completions/min_length": 33.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.6512926816940308, "rewards/meter/std": 0.36039817333221436, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9018807411193848, "rewards/repeat_soft/std": 0.051085714250802994, "rewards/judge_quality/mean": 0.42249998450279236, "rewards/judge_quality/std": 0.14616528153419495, "rewards/total_composite/mean": 0.6600198149681091, "rewards/total_composite/std": 0.13950856029987335, "reward": 0.6600198149681091, "reward_std": 0.13950856029987335, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1262923926115036, "sampling/sampling_logp_difference/max": 1.4784135818481445, "sampling/importance_sampling_ratio/min": 0.22799910604953766, "sampling/importance_sampling_ratio/mean": 1.0322117805480957, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8790705278515816, "clip_ratio/low_mean": 0.06486461963504553, "clip_ratio/low_min": 0.06486461963504553, "clip_ratio/high_mean": 0.06296138186007738, "clip_ratio/high_max": 0.06296138186007738, "clip_ratio/region_mean": 0.1278260014951229, "reward_total_mean": 0.6600198149681091, "reward_meter_mean": 0.6512926816940308, "reward_meter_std": 0.36039817333221436, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9018807411193848, "reward_repeat_soft_std": 0.051085714250802994, "reward_judge_quality_mean": 0.42249998450279236, "reward_judge_quality_std": 0.14616528153419495, "reward_total_composite_mean": 0.6600198149681091, "reward_total_composite_std": 0.13950856029987335} {"timestamp_utc": "2026-04-13T00:21:08Z", "mode": "train", "global_step": 847, "epoch": 0.08508287292817679, "loss": 0.0295, "grad_norm": 13.948565483093262, "learning_rate": 7.4363636363636375e-06, "num_tokens": 1548262.0, "completions/mean_length": 39.5, "completions/min_length": 28.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.5, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.6500986814498901, "rewards/meter/std": 0.39183011651039124, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8989070653915405, "rewards/repeat_soft/std": 0.1315387636423111, "rewards/judge_quality/mean": 0.5637500286102295, "rewards/judge_quality/std": 0.22012579441070557, "rewards/total_composite/mean": 0.7015601396560669, "rewards/total_composite/std": 0.12817999720573425, "reward": 0.7015601396560669, "reward_std": 0.12817999720573425, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1463286578655243, "sampling/sampling_logp_difference/max": 1.708939552307129, "sampling/importance_sampling_ratio/min": 0.18105769157409668, "sampling/importance_sampling_ratio/mean": 0.9976785778999329, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7477677837014198, "clip_ratio/low_mean": 0.08366481121629477, "clip_ratio/low_min": 0.08366481121629477, "clip_ratio/high_mean": 0.07913488056510687, "clip_ratio/high_max": 0.07913488056510687, "clip_ratio/region_mean": 0.16279969178140163, "reward_total_mean": 0.7015601396560669, "reward_meter_mean": 0.6500986814498901, "reward_meter_std": 0.39183011651039124, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8989070653915405, "reward_repeat_soft_std": 0.1315387636423111, "reward_judge_quality_mean": 0.5637500286102295, "reward_judge_quality_std": 0.22012579441070557, "reward_total_composite_mean": 0.7015601396560669, "reward_total_composite_std": 0.12817999720573425} {"timestamp_utc": "2026-04-13T00:21:18Z", "mode": "train", "global_step": 848, "epoch": 0.08518332496233048, "loss": 0.074, "grad_norm": 20.567232131958008, "learning_rate": 7.433333333333334e-06, "num_tokens": 1549799.0, "completions/mean_length": 33.125, "completions/min_length": 27.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.125, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.6442090272903442, "rewards/meter/std": 0.41661331057548523, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8340409994125366, "rewards/repeat_soft/std": 0.21218815445899963, "rewards/judge_quality/mean": 0.5149999856948853, "rewards/judge_quality/std": 0.2676885426044464, "rewards/total_composite/mean": 0.6777981519699097, "rewards/total_composite/std": 0.2188120186328888, "reward": 0.6777981519699097, "reward_std": 0.2188120186328888, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1717747300863266, "sampling/sampling_logp_difference/max": 3.7546844482421875, "sampling/importance_sampling_ratio/min": 0.023407835513353348, "sampling/importance_sampling_ratio/mean": 1.009258508682251, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8038934953510761, "clip_ratio/low_mean": 0.0711048454977572, "clip_ratio/low_min": 0.0711048454977572, "clip_ratio/high_mean": 0.08729672990739346, "clip_ratio/high_max": 0.08729672990739346, "clip_ratio/region_mean": 0.15840157540515065, "reward_total_mean": 0.6777981519699097, "reward_meter_mean": 0.6442090272903442, "reward_meter_std": 0.41661331057548523, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8340409994125366, "reward_repeat_soft_std": 0.21218815445899963, "reward_judge_quality_mean": 0.5149999856948853, "reward_judge_quality_std": 0.2676885426044464, "reward_total_composite_mean": 0.6777981519699097, "reward_total_composite_std": 0.2188120186328888} {"timestamp_utc": "2026-04-13T00:21:25Z", "mode": "train", "global_step": 849, "epoch": 0.08528377699648418, "loss": 0.0779, "grad_norm": 13.574016571044922, "learning_rate": 7.430303030303031e-06, "num_tokens": 1551287.0, "completions/mean_length": 32.0, "completions/min_length": 28.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9146527051925659, "rewards/meter/std": 0.14170169830322266, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8450944423675537, "rewards/repeat_soft/std": 0.0914902463555336, "rewards/judge_quality/mean": 0.4300000071525574, "rewards/judge_quality/std": 0.21993505954742432, "rewards/total_composite/mean": 0.7751032114028931, "rewards/total_composite/std": 0.11489620804786682, "reward": 0.7751032114028931, "reward_std": 0.11489620804786682, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11597013473510742, "sampling/sampling_logp_difference/max": 0.947547197341919, "sampling/importance_sampling_ratio/min": 0.38769078254699707, "sampling/importance_sampling_ratio/mean": 0.9859033823013306, "sampling/importance_sampling_ratio/max": 1.7388297319412231, "entropy": 0.7218378856778145, "clip_ratio/low_mean": 0.030964053235948086, "clip_ratio/low_min": 0.030964053235948086, "clip_ratio/high_mean": 0.10120400600135326, "clip_ratio/high_max": 0.10120400600135326, "clip_ratio/region_mean": 0.13216805923730135, "reward_total_mean": 0.7751032114028931, "reward_meter_mean": 0.9146527051925659, "reward_meter_std": 0.14170169830322266, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8450944423675537, "reward_repeat_soft_std": 0.0914902463555336, "reward_judge_quality_mean": 0.4300000071525574, "reward_judge_quality_std": 0.21993505954742432, "reward_total_composite_mean": 0.7751032114028931, "reward_total_composite_std": 0.11489620804786682} {"timestamp_utc": "2026-04-13T00:21:31Z", "mode": "train", "global_step": 850, "epoch": 0.08538422903063787, "loss": -0.0085, "grad_norm": 20.61587142944336, "learning_rate": 7.4272727272727275e-06, "num_tokens": 1552731.0, "completions/mean_length": 32.5, "completions/min_length": 29.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.5, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.7259894609451294, "rewards/meter/std": 0.28846290707588196, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9292584657669067, "rewards/repeat_soft/std": 0.09010230749845505, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.6914961338043213, "rewards/total_composite/std": 0.14543992280960083, "reward": 0.6914961338043213, "reward_std": 0.14543990790843964, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19902531802654266, "sampling/sampling_logp_difference/max": 2.0006179809570312, "sampling/importance_sampling_ratio/min": 0.13525167107582092, "sampling/importance_sampling_ratio/mean": 0.9891232848167419, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.089644156396389, "clip_ratio/low_mean": 0.048766305670142174, "clip_ratio/low_min": 0.048766305670142174, "clip_ratio/high_mean": 0.10142362583428621, "clip_ratio/high_max": 0.10142362583428621, "clip_ratio/region_mean": 0.1501899315044284, "reward_total_mean": 0.6914961338043213, "reward_meter_mean": 0.7259894609451294, "reward_meter_std": 0.28846290707588196, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9292584657669067, "reward_repeat_soft_std": 0.09010230749845505, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.6914961338043213, "reward_total_composite_std": 0.14543992280960083} {"timestamp_utc": "2026-04-13T00:22:15Z", "mode": "eval", "global_step": 850, "epoch": 0.08538422903063787, "eval_loss": NaN, "eval_runtime": 43.7337, "eval_samples_per_second": 1.829, "eval_steps_per_second": 0.229, "eval_num_tokens": 1552731.0, "eval_completions/mean_length": 51.8875, "eval_completions/min_length": 26.9, "eval_completions/max_length": 94.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 51.8875, "eval_completions/min_terminated_length": 26.9, "eval_completions/max_terminated_length": 94.0, "eval_rewards/meter/mean": 0.8370405733585358, "eval_rewards/meter/std": 0.23556316308677197, "eval_rewards/count_adherence/mean": 0.8356249928474426, "eval_rewards/count_adherence/std": 0.15408377796411515, "eval_rewards/hard_gate/mean": 0.9875, "eval_rewards/hard_gate/std": 0.03535533845424652, "eval_rewards/repeat_soft/mean": 0.8487918913364411, "eval_rewards/repeat_soft/std": 0.16261641010642053, "eval_rewards/judge_quality/mean": 0.389875003695488, "eval_rewards/judge_quality/std": 0.18367089554667473, "eval_rewards/total_composite/mean": 0.6954112112522125, "eval_rewards/total_composite/std": 0.15140126794576644, "eval_reward": 0.6954112112522125, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.07251444272696972, "eval_sampling/sampling_logp_difference/max": 1.0466674327850343, "eval_sampling/importance_sampling_ratio/min": 0.35695213079452515, "eval_sampling/importance_sampling_ratio/mean": 1.0176384687423705, "eval_sampling/importance_sampling_ratio/max": 1.4075155138969422, "eval_entropy": 0.835816067457199, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6954112112522125, "eval_reward_meter_mean": 0.8370405733585358, "eval_reward_meter_std": 0.23556316308677197, "eval_reward_count_adherence_mean": 0.8356249928474426, "eval_reward_count_adherence_std": 0.15408377796411515, "eval_reward_hard_gate_mean": 0.9875, "eval_reward_hard_gate_std": 0.03535533845424652, "eval_reward_repeat_soft_mean": 0.8487918913364411, "eval_reward_repeat_soft_std": 0.16261641010642053, "eval_reward_judge_quality_mean": 0.389875003695488, "eval_reward_judge_quality_std": 0.18367089554667473, "eval_reward_total_composite_mean": 0.6954112112522125, "eval_reward_total_composite_std": 0.15140126794576644} {"timestamp_utc": "2026-04-13T00:22:25Z", "mode": "train", "global_step": 851, "epoch": 0.08548468106479157, "loss": 0.025, "grad_norm": 14.569260597229004, "learning_rate": 7.424242424242425e-06, "num_tokens": 1554329.0, "completions/mean_length": 37.75, "completions/min_length": 32.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.75, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.727995753288269, "rewards/meter/std": 0.27383625507354736, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9678243398666382, "rewards/repeat_soft/std": 0.04353925585746765, "rewards/judge_quality/mean": 0.7987500429153442, "rewards/judge_quality/std": 0.22465451061725616, "rewards/total_composite/mean": 0.8140054941177368, "rewards/total_composite/std": 0.1546793282032013, "reward": 0.8140054941177368, "reward_std": 0.1546793133020401, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14232036471366882, "sampling/sampling_logp_difference/max": 1.4657387733459473, "sampling/importance_sampling_ratio/min": 0.2309073507785797, "sampling/importance_sampling_ratio/mean": 1.034467339515686, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7994279712438583, "clip_ratio/low_mean": 0.06643260829150677, "clip_ratio/low_min": 0.06643260829150677, "clip_ratio/high_mean": 0.06748099531978369, "clip_ratio/high_max": 0.06748099531978369, "clip_ratio/region_mean": 0.13391360361129045, "reward_total_mean": 0.8140054941177368, "reward_meter_mean": 0.727995753288269, "reward_meter_std": 0.27383625507354736, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9678243398666382, "reward_repeat_soft_std": 0.04353925585746765, "reward_judge_quality_mean": 0.7987500429153442, "reward_judge_quality_std": 0.22465451061725616, "reward_total_composite_mean": 0.8140054941177368, "reward_total_composite_std": 0.1546793282032013} {"timestamp_utc": "2026-04-13T00:22:32Z", "mode": "train", "global_step": 852, "epoch": 0.08558513309894525, "loss": 0.0483, "grad_norm": 17.19428253173828, "learning_rate": 7.421212121212121e-06, "num_tokens": 1556225.0, "completions/mean_length": 58.0, "completions/min_length": 51.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.8533411026000977, "rewards/meter/std": 0.1589355319738388, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9935046434402466, "rewards/repeat_soft/std": 0.005225415341556072, "rewards/judge_quality/mean": 0.7200000286102295, "rewards/judge_quality/std": 0.20701968669891357, "rewards/total_composite/mean": 0.8493539094924927, "rewards/total_composite/std": 0.1105462983250618, "reward": 0.8493539094924927, "reward_std": 0.1105462983250618, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13720232248306274, "sampling/sampling_logp_difference/max": 1.713212490081787, "sampling/importance_sampling_ratio/min": 0.18028569221496582, "sampling/importance_sampling_ratio/mean": 1.0133371353149414, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6849218234419823, "clip_ratio/low_mean": 0.06663383170962334, "clip_ratio/low_min": 0.06663383170962334, "clip_ratio/high_mean": 0.057162309531122446, "clip_ratio/high_max": 0.057162309531122446, "clip_ratio/region_mean": 0.12379614124074578, "reward_total_mean": 0.8493539094924927, "reward_meter_mean": 0.8533411026000977, "reward_meter_std": 0.1589355319738388, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9935046434402466, "reward_repeat_soft_std": 0.005225415341556072, "reward_judge_quality_mean": 0.7200000286102295, "reward_judge_quality_std": 0.20701968669891357, "reward_total_composite_mean": 0.8493539094924927, "reward_total_composite_std": 0.1105462983250618} {"timestamp_utc": "2026-04-13T00:22:39Z", "mode": "train", "global_step": 853, "epoch": 0.08568558513309894, "loss": 0.0519, "grad_norm": 14.495819091796875, "learning_rate": 7.4181818181818185e-06, "num_tokens": 1557678.0, "completions/mean_length": 34.625, "completions/min_length": 29.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.625, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.4830441474914551, "rewards/meter/std": 0.3940315544605255, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9944974184036255, "rewards/repeat_soft/std": 0.009725687094032764, "rewards/judge_quality/mean": 0.7362500429153442, "rewards/judge_quality/std": 0.25376805663108826, "rewards/total_composite/mean": 0.6876946091651917, "rewards/total_composite/std": 0.23385725915431976, "reward": 0.6876946091651917, "reward_std": 0.23385725915431976, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1623799055814743, "sampling/sampling_logp_difference/max": 1.2787697315216064, "sampling/importance_sampling_ratio/min": 0.27837955951690674, "sampling/importance_sampling_ratio/mean": 1.0178101062774658, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9765953943133354, "clip_ratio/low_mean": 0.04243775736540556, "clip_ratio/low_min": 0.04243775736540556, "clip_ratio/high_mean": 0.06806604936718941, "clip_ratio/high_max": 0.06806604936718941, "clip_ratio/region_mean": 0.11050380673259497, "reward_total_mean": 0.6876946091651917, "reward_meter_mean": 0.4830441474914551, "reward_meter_std": 0.3940315544605255, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9944974184036255, "reward_repeat_soft_std": 0.009725687094032764, "reward_judge_quality_mean": 0.7362500429153442, "reward_judge_quality_std": 0.25376805663108826, "reward_total_composite_mean": 0.6876946091651917, "reward_total_composite_std": 0.23385725915431976} {"timestamp_utc": "2026-04-13T00:22:45Z", "mode": "train", "global_step": 854, "epoch": 0.08578603716725264, "loss": 0.0452, "grad_norm": 18.672758102416992, "learning_rate": 7.415151515151515e-06, "num_tokens": 1559126.0, "completions/mean_length": 28.0, "completions/min_length": 20.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.0, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9955481290817261, "rewards/meter/std": 0.005081599112600088, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9200875759124756, "rewards/repeat_soft/std": 0.11295874416828156, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.1297180950641632, "rewards/total_composite/mean": 0.8021304607391357, "rewards/total_composite/std": 0.04810316860675812, "reward": 0.8021304607391357, "reward_std": 0.04810316860675812, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14328992366790771, "sampling/sampling_logp_difference/max": 0.7274255752563477, "sampling/importance_sampling_ratio/min": 0.48315125703811646, "sampling/importance_sampling_ratio/mean": 1.0338995456695557, "sampling/importance_sampling_ratio/max": 1.795361042022705, "entropy": 1.3230515271425247, "clip_ratio/low_mean": 0.035317461006343365, "clip_ratio/low_min": 0.035317461006343365, "clip_ratio/high_mean": 0.08447418361902237, "clip_ratio/high_max": 0.08447418361902237, "clip_ratio/region_mean": 0.11979164462536573, "reward_total_mean": 0.8021304607391357, "reward_meter_mean": 0.9955481290817261, "reward_meter_std": 0.005081599112600088, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9200875759124756, "reward_repeat_soft_std": 0.11295874416828156, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.1297180950641632, "reward_total_composite_mean": 0.8021304607391357, "reward_total_composite_std": 0.04810316860675812} {"timestamp_utc": "2026-04-13T00:22:53Z", "mode": "train", "global_step": 855, "epoch": 0.08588648920140633, "loss": 0.0494, "grad_norm": 17.159086227416992, "learning_rate": 7.412121212121213e-06, "num_tokens": 1560718.0, "completions/mean_length": 44.0, "completions/min_length": 33.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9538765549659729, "rewards/meter/std": 0.06620649993419647, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9629954099655151, "rewards/repeat_soft/std": 0.041554927825927734, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.813918948173523, "rewards/total_composite/std": 0.06917697191238403, "reward": 0.813918948173523, "reward_std": 0.06917694956064224, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18290437757968903, "sampling/sampling_logp_difference/max": 1.4870657920837402, "sampling/importance_sampling_ratio/min": 0.22603492438793182, "sampling/importance_sampling_ratio/mean": 1.0220959186553955, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2898493483662605, "clip_ratio/low_mean": 0.09779283311218023, "clip_ratio/low_min": 0.09779283311218023, "clip_ratio/high_mean": 0.09435221180319786, "clip_ratio/high_max": 0.09435221180319786, "clip_ratio/region_mean": 0.1921450449153781, "reward_total_mean": 0.813918948173523, "reward_meter_mean": 0.9538765549659729, "reward_meter_std": 0.06620649993419647, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9629954099655151, "reward_repeat_soft_std": 0.041554927825927734, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.813918948173523, "reward_total_composite_std": 0.06917697191238403} {"timestamp_utc": "2026-04-13T00:23:00Z", "mode": "train", "global_step": 856, "epoch": 0.08598694123556001, "loss": 0.1128, "grad_norm": 24.1082763671875, "learning_rate": 7.40909090909091e-06, "num_tokens": 1562267.0, "completions/mean_length": 28.625, "completions/min_length": 26.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.625, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.6248811483383179, "rewards/meter/std": 0.309164822101593, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8668676018714905, "rewards/repeat_soft/std": 0.16119304299354553, "rewards/judge_quality/mean": 0.4712499976158142, "rewards/judge_quality/std": 0.30572807788848877, "rewards/total_composite/mean": 0.659258246421814, "rewards/total_composite/std": 0.16146840155124664, "reward": 0.659258246421814, "reward_std": 0.16146838665008545, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17463691532611847, "sampling/sampling_logp_difference/max": 1.3818840980529785, "sampling/importance_sampling_ratio/min": 0.25110501050949097, "sampling/importance_sampling_ratio/mean": 1.0166670083999634, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0091931372880936, "clip_ratio/low_mean": 0.07936622109264135, "clip_ratio/low_min": 0.07936622109264135, "clip_ratio/high_mean": 0.04223901126533747, "clip_ratio/high_max": 0.04223901126533747, "clip_ratio/region_mean": 0.12160523235797882, "reward_total_mean": 0.659258246421814, "reward_meter_mean": 0.6248811483383179, "reward_meter_std": 0.309164822101593, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8668676018714905, "reward_repeat_soft_std": 0.16119304299354553, "reward_judge_quality_mean": 0.4712499976158142, "reward_judge_quality_std": 0.30572807788848877, "reward_total_composite_mean": 0.659258246421814, "reward_total_composite_std": 0.16146840155124664} {"timestamp_utc": "2026-04-13T00:23:07Z", "mode": "train", "global_step": 857, "epoch": 0.08608739326971371, "loss": -0.0378, "grad_norm": 8.960577011108398, "learning_rate": 7.406060606060607e-06, "num_tokens": 1564311.0, "completions/mean_length": 64.5, "completions/min_length": 49.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.5, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9750339388847351, "rewards/meter/std": 0.02742379531264305, "rewards/count_adherence/mean": 0.7749999761581421, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8541339635848999, "rewards/repeat_soft/std": 0.17220349609851837, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7536786794662476, "rewards/total_composite/std": 0.046955011785030365, "reward": 0.7536786794662476, "reward_std": 0.046955011785030365, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13840796053409576, "sampling/sampling_logp_difference/max": 2.025650978088379, "sampling/importance_sampling_ratio/min": 0.13190795481204987, "sampling/importance_sampling_ratio/mean": 1.017824649810791, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1017243042588234, "clip_ratio/low_mean": 0.038179040187969804, "clip_ratio/low_min": 0.038179040187969804, "clip_ratio/high_mean": 0.09566516242921352, "clip_ratio/high_max": 0.09566516242921352, "clip_ratio/region_mean": 0.13384420261718333, "reward_total_mean": 0.7536786794662476, "reward_meter_mean": 0.9750339388847351, "reward_meter_std": 0.02742379531264305, "reward_count_adherence_mean": 0.7749999761581421, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8541339635848999, "reward_repeat_soft_std": 0.17220349609851837, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7536786794662476, "reward_total_composite_std": 0.046955011785030365} {"timestamp_utc": "2026-04-13T00:23:14Z", "mode": "train", "global_step": 858, "epoch": 0.0861878453038674, "loss": 0.0251, "grad_norm": 11.00955581665039, "learning_rate": 7.403030303030304e-06, "num_tokens": 1565794.0, "completions/mean_length": 40.375, "completions/min_length": 37.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.375, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.968699038028717, "rewards/meter/std": 0.04381776228547096, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9832814931869507, "rewards/repeat_soft/std": 0.02504497952759266, "rewards/judge_quality/mean": 0.5349999666213989, "rewards/judge_quality/std": 0.24951381981372833, "rewards/total_composite/mean": 0.8447426557540894, "rewards/total_composite/std": 0.07388333976268768, "reward": 0.8447426557540894, "reward_std": 0.07388332486152649, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12890420854091644, "sampling/sampling_logp_difference/max": 2.447132110595703, "sampling/importance_sampling_ratio/min": 0.08654142916202545, "sampling/importance_sampling_ratio/mean": 0.9985030293464661, "sampling/importance_sampling_ratio/max": 1.9749053716659546, "entropy": 0.7274476699531078, "clip_ratio/low_mean": 0.05627789627760649, "clip_ratio/low_min": 0.05627789627760649, "clip_ratio/high_mean": 0.03407012112438679, "clip_ratio/high_max": 0.03407012112438679, "clip_ratio/region_mean": 0.09034801740199327, "reward_total_mean": 0.8447426557540894, "reward_meter_mean": 0.968699038028717, "reward_meter_std": 0.04381776228547096, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9832814931869507, "reward_repeat_soft_std": 0.02504497952759266, "reward_judge_quality_mean": 0.5349999666213989, "reward_judge_quality_std": 0.24951381981372833, "reward_total_composite_mean": 0.8447426557540894, "reward_total_composite_std": 0.07388333976268768} {"timestamp_utc": "2026-04-13T00:23:20Z", "mode": "train", "global_step": 859, "epoch": 0.0862882973380211, "loss": 0.1242, "grad_norm": 23.484825134277344, "learning_rate": 7.4e-06, "num_tokens": 1567209.0, "completions/mean_length": 23.875, "completions/min_length": 20.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.875, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.40358707308769226, "rewards/meter/std": 0.323246568441391, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9907225966453552, "rewards/repeat_soft/std": 0.024169057607650757, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.6339364051818848, "rewards/total_composite/std": 0.16726504266262054, "reward": 0.6339364051818848, "reward_std": 0.16726504266262054, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16398359835147858, "sampling/sampling_logp_difference/max": 2.7904655933380127, "sampling/importance_sampling_ratio/min": 0.06139262020587921, "sampling/importance_sampling_ratio/mean": 1.0259705781936646, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7785592041909695, "clip_ratio/low_mean": 0.06666666734963655, "clip_ratio/low_min": 0.06666666734963655, "clip_ratio/high_mean": 0.07523435773327947, "clip_ratio/high_max": 0.07523435773327947, "clip_ratio/region_mean": 0.14190102508291602, "reward_total_mean": 0.6339364051818848, "reward_meter_mean": 0.40358707308769226, "reward_meter_std": 0.323246568441391, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9907225966453552, "reward_repeat_soft_std": 0.024169057607650757, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.6339364051818848, "reward_total_composite_std": 0.16726504266262054} {"timestamp_utc": "2026-04-13T00:23:28Z", "mode": "train", "global_step": 860, "epoch": 0.08638874937217479, "loss": 0.0132, "grad_norm": 12.651262283325195, "learning_rate": 7.396969696969698e-06, "num_tokens": 1569106.0, "completions/mean_length": 62.125, "completions/min_length": 45.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.125, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.9799870848655701, "rewards/meter/std": 0.03022618778049946, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8651416301727295, "rewards/repeat_soft/std": 0.07978132367134094, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.7367583513259888, "rewards/total_composite/std": 0.03532982990145683, "reward": 0.7367583513259888, "reward_std": 0.03532983735203743, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15493705868721008, "sampling/sampling_logp_difference/max": 2.7464537620544434, "sampling/importance_sampling_ratio/min": 0.0641549676656723, "sampling/importance_sampling_ratio/mean": 1.0149576663970947, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9685847759246826, "clip_ratio/low_mean": 0.05728174652904272, "clip_ratio/low_min": 0.05728174652904272, "clip_ratio/high_mean": 0.0680325711145997, "clip_ratio/high_max": 0.0680325711145997, "clip_ratio/region_mean": 0.12531431764364243, "reward_total_mean": 0.7367583513259888, "reward_meter_mean": 0.9799870848655701, "reward_meter_std": 0.03022618778049946, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8651416301727295, "reward_repeat_soft_std": 0.07978132367134094, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.7367583513259888, "reward_total_composite_std": 0.03532982990145683} {"timestamp_utc": "2026-04-13T00:23:35Z", "mode": "train", "global_step": 861, "epoch": 0.08648920140632847, "loss": 0.0578, "grad_norm": 5.290957927703857, "learning_rate": 7.393939393939395e-06, "num_tokens": 1571372.0, "completions/mean_length": 106.25, "completions/min_length": 86.0, "completions/max_length": 144.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.25, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 144.0, "rewards/meter/mean": 0.9874061942100525, "rewards/meter/std": 0.005764068104326725, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.31970518827438354, "rewards/repeat_soft/std": 0.09105415642261505, "rewards/judge_quality/mean": 0.23375000059604645, "rewards/judge_quality/std": 0.09006941318511963, "rewards/total_composite/mean": 0.6714283227920532, "rewards/total_composite/std": 0.030555443838238716, "reward": 0.6714283227920532, "reward_std": 0.030555451288819313, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.060868434607982635, "sampling/sampling_logp_difference/max": 1.256934642791748, "sampling/importance_sampling_ratio/min": 0.2845248579978943, "sampling/importance_sampling_ratio/mean": 1.0136464834213257, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40283073857426643, "clip_ratio/low_mean": 0.02639946830458939, "clip_ratio/low_min": 0.02639946830458939, "clip_ratio/high_mean": 0.018980397377163172, "clip_ratio/high_max": 0.018980397377163172, "clip_ratio/region_mean": 0.04537986568175256, "reward_total_mean": 0.6714283227920532, "reward_meter_mean": 0.9874061942100525, "reward_meter_std": 0.005764068104326725, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.31970518827438354, "reward_repeat_soft_std": 0.09105415642261505, "reward_judge_quality_mean": 0.23375000059604645, "reward_judge_quality_std": 0.09006941318511963, "reward_total_composite_mean": 0.6714283227920532, "reward_total_composite_std": 0.030555443838238716} {"timestamp_utc": "2026-04-13T00:23:42Z", "mode": "train", "global_step": 862, "epoch": 0.08658965344048217, "loss": -0.0756, "grad_norm": 21.948152542114258, "learning_rate": 7.390909090909092e-06, "num_tokens": 1573018.0, "completions/mean_length": 41.75, "completions/min_length": 28.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.75, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.8240319490432739, "rewards/meter/std": 0.3177677094936371, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9459333419799805, "rewards/repeat_soft/std": 0.06348675489425659, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.1954299360513687, "rewards/total_composite/mean": 0.7674077749252319, "rewards/total_composite/std": 0.16564927995204926, "reward": 0.7674077749252319, "reward_std": 0.16564927995204926, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17575190961360931, "sampling/sampling_logp_difference/max": 2.0518672466278076, "sampling/importance_sampling_ratio/min": 0.1284947395324707, "sampling/importance_sampling_ratio/mean": 1.0029107332229614, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7612264752388, "clip_ratio/low_mean": 0.045463320799171925, "clip_ratio/low_min": 0.045463320799171925, "clip_ratio/high_mean": 0.08799278549849987, "clip_ratio/high_max": 0.08799278549849987, "clip_ratio/region_mean": 0.1334561062976718, "reward_total_mean": 0.7674077749252319, "reward_meter_mean": 0.8240319490432739, "reward_meter_std": 0.3177677094936371, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9459333419799805, "reward_repeat_soft_std": 0.06348675489425659, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.1954299360513687, "reward_total_composite_mean": 0.7674077749252319, "reward_total_composite_std": 0.16564927995204926} {"timestamp_utc": "2026-04-13T00:23:53Z", "mode": "train", "global_step": 863, "epoch": 0.08669010547463586, "loss": -0.1228, "grad_norm": 3.5906898975372314, "learning_rate": 7.3878787878787885e-06, "num_tokens": 1574637.0, "completions/mean_length": 102.375, "completions/min_length": 38.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 43.85714340209961, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.3147362470626831, "rewards/meter/std": 0.31491291522979736, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9720009565353394, "rewards/repeat_soft/std": 0.04115351662039757, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.18845234811306, "rewards/total_composite/mean": 0.47026723623275757, "rewards/total_composite/std": 0.22440779209136963, "reward": 0.47026723623275757, "reward_std": 0.22440777719020844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18055635690689087, "sampling/sampling_logp_difference/max": 1.8157918453216553, "sampling/importance_sampling_ratio/min": 0.16270901262760162, "sampling/importance_sampling_ratio/mean": 1.039677619934082, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2936688587069511, "clip_ratio/low_mean": 0.05324357561767101, "clip_ratio/low_min": 0.05324357561767101, "clip_ratio/high_mean": 0.07518970035016537, "clip_ratio/high_max": 0.07518970035016537, "clip_ratio/region_mean": 0.12843327596783638, "reward_total_mean": 0.47026723623275757, "reward_meter_mean": 0.3147362470626831, "reward_meter_std": 0.31491291522979736, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9720009565353394, "reward_repeat_soft_std": 0.04115351662039757, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.18845234811306, "reward_total_composite_mean": 0.47026723623275757, "reward_total_composite_std": 0.22440779209136963} {"timestamp_utc": "2026-04-13T00:23:59Z", "mode": "train", "global_step": 864, "epoch": 0.08679055750878956, "loss": -0.0449, "grad_norm": 22.377986907958984, "learning_rate": 7.384848484848486e-06, "num_tokens": 1576370.0, "completions/mean_length": 28.625, "completions/min_length": 22.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.625, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.6938362121582031, "rewards/meter/std": 0.2758866250514984, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9893450736999512, "rewards/repeat_soft/std": 0.01325307972729206, "rewards/judge_quality/mean": 0.8237500190734863, "rewards/judge_quality/std": 0.2722361385822296, "rewards/total_composite/mean": 0.8082858324050903, "rewards/total_composite/std": 0.17534098029136658, "reward": 0.8082858324050903, "reward_std": 0.17534098029136658, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.144114688038826, "sampling/sampling_logp_difference/max": 1.4582843780517578, "sampling/importance_sampling_ratio/min": 0.23263505101203918, "sampling/importance_sampling_ratio/mean": 1.024067997932434, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.856540709733963, "clip_ratio/low_mean": 0.07460172846913338, "clip_ratio/low_min": 0.07460172846913338, "clip_ratio/high_mean": 0.04298029653728008, "clip_ratio/high_max": 0.04298029653728008, "clip_ratio/region_mean": 0.11758202500641346, "reward_total_mean": 0.8082858324050903, "reward_meter_mean": 0.6938362121582031, "reward_meter_std": 0.2758866250514984, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9893450736999512, "reward_repeat_soft_std": 0.01325307972729206, "reward_judge_quality_mean": 0.8237500190734863, "reward_judge_quality_std": 0.2722361385822296, "reward_total_composite_mean": 0.8082858324050903, "reward_total_composite_std": 0.17534098029136658} {"timestamp_utc": "2026-04-13T00:24:11Z", "mode": "train", "global_step": 865, "epoch": 0.08689100954294325, "loss": 0.0203, "grad_norm": 13.628290176391602, "learning_rate": 7.381818181818182e-06, "num_tokens": 1577944.0, "completions/mean_length": 42.75, "completions/min_length": 35.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.75, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.722493052482605, "rewards/meter/std": 0.32507190108299255, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.996993362903595, "rewards/repeat_soft/std": 0.005423650611191988, "rewards/judge_quality/mean": 0.6187499761581421, "rewards/judge_quality/std": 0.24976776540279388, "rewards/total_composite/mean": 0.7604461908340454, "rewards/total_composite/std": 0.17658939957618713, "reward": 0.7604461908340454, "reward_std": 0.17658939957618713, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17203688621520996, "sampling/sampling_logp_difference/max": 1.912567138671875, "sampling/importance_sampling_ratio/min": 0.1477007418870926, "sampling/importance_sampling_ratio/mean": 0.9954968094825745, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9498367160558701, "clip_ratio/low_mean": 0.057002972811460495, "clip_ratio/low_min": 0.057002972811460495, "clip_ratio/high_mean": 0.12667176872491837, "clip_ratio/high_max": 0.12667176872491837, "clip_ratio/region_mean": 0.18367474153637886, "reward_total_mean": 0.7604461908340454, "reward_meter_mean": 0.722493052482605, "reward_meter_std": 0.32507190108299255, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.996993362903595, "reward_repeat_soft_std": 0.005423650611191988, "reward_judge_quality_mean": 0.6187499761581421, "reward_judge_quality_std": 0.24976776540279388, "reward_total_composite_mean": 0.7604461908340454, "reward_total_composite_std": 0.17658939957618713} {"timestamp_utc": "2026-04-13T00:24:17Z", "mode": "train", "global_step": 866, "epoch": 0.08699146157709693, "loss": 0.0363, "grad_norm": 18.01833724975586, "learning_rate": 7.378787878787879e-06, "num_tokens": 1579813.0, "completions/mean_length": 37.625, "completions/min_length": 33.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.625, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9021809101104736, "rewards/meter/std": 0.1569957584142685, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7428385615348816, "rewards/repeat_soft/std": 0.18590037524700165, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.814765214920044, "rewards/total_composite/std": 0.1331525444984436, "reward": 0.814765214920044, "reward_std": 0.1331525295972824, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1331234574317932, "sampling/sampling_logp_difference/max": 2.2153072357177734, "sampling/importance_sampling_ratio/min": 0.10911998152732849, "sampling/importance_sampling_ratio/mean": 1.0110489130020142, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8297382295131683, "clip_ratio/low_mean": 0.06638337299227715, "clip_ratio/low_min": 0.06638337299227715, "clip_ratio/high_mean": 0.071092639118433, "clip_ratio/high_max": 0.071092639118433, "clip_ratio/region_mean": 0.13747601211071014, "reward_total_mean": 0.814765214920044, "reward_meter_mean": 0.9021809101104736, "reward_meter_std": 0.1569957584142685, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7428385615348816, "reward_repeat_soft_std": 0.18590037524700165, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.814765214920044, "reward_total_composite_std": 0.1331525444984436} {"timestamp_utc": "2026-04-13T00:24:24Z", "mode": "train", "global_step": 867, "epoch": 0.08709191361125063, "loss": 0.0299, "grad_norm": 18.87558364868164, "learning_rate": 7.375757575757576e-06, "num_tokens": 1581419.0, "completions/mean_length": 34.75, "completions/min_length": 29.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.75, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.886028528213501, "rewards/meter/std": 0.1902603805065155, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.11785111576318741, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9123224020004272, "rewards/repeat_soft/std": 0.08659925311803818, "rewards/judge_quality/mean": 0.48124998807907104, "rewards/judge_quality/std": 0.1606627106666565, "rewards/total_composite/mean": 0.740570068359375, "rewards/total_composite/std": 0.11736400425434113, "reward": 0.740570068359375, "reward_std": 0.11736401170492172, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16493307054042816, "sampling/sampling_logp_difference/max": 1.5325326919555664, "sampling/importance_sampling_ratio/min": 0.2159879505634308, "sampling/importance_sampling_ratio/mean": 1.0289409160614014, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0333832427859306, "clip_ratio/low_mean": 0.014285714365541935, "clip_ratio/low_min": 0.014285714365541935, "clip_ratio/high_mean": 0.14088684413582087, "clip_ratio/high_max": 0.14088684413582087, "clip_ratio/region_mean": 0.1551725585013628, "reward_total_mean": 0.740570068359375, "reward_meter_mean": 0.886028528213501, "reward_meter_std": 0.1902603805065155, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.11785111576318741, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9123224020004272, "reward_repeat_soft_std": 0.08659925311803818, "reward_judge_quality_mean": 0.48124998807907104, "reward_judge_quality_std": 0.1606627106666565, "reward_total_composite_mean": 0.740570068359375, "reward_total_composite_std": 0.11736400425434113} {"timestamp_utc": "2026-04-13T00:24:31Z", "mode": "train", "global_step": 868, "epoch": 0.08719236564540432, "loss": 0.1551, "grad_norm": 10.924677848815918, "learning_rate": 7.372727272727274e-06, "num_tokens": 1583077.0, "completions/mean_length": 45.25, "completions/min_length": 37.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.25, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9903303384780884, "rewards/meter/std": 0.00382985919713974, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9231892228126526, "rewards/repeat_soft/std": 0.049479253590106964, "rewards/judge_quality/mean": 0.4437499940395355, "rewards/judge_quality/std": 0.2084595113992691, "rewards/total_composite/mean": 0.8210926055908203, "rewards/total_composite/std": 0.0651756301522255, "reward": 0.8210926055908203, "reward_std": 0.0651756152510643, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1597650647163391, "sampling/sampling_logp_difference/max": 1.4619789123535156, "sampling/importance_sampling_ratio/min": 0.231777161359787, "sampling/importance_sampling_ratio/mean": 1.001970648765564, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0997956618666649, "clip_ratio/low_mean": 0.08201638050377369, "clip_ratio/low_min": 0.08201638050377369, "clip_ratio/high_mean": 0.03813813906162977, "clip_ratio/high_max": 0.03813813906162977, "clip_ratio/region_mean": 0.12015451956540346, "reward_total_mean": 0.8210926055908203, "reward_meter_mean": 0.9903303384780884, "reward_meter_std": 0.00382985919713974, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9231892228126526, "reward_repeat_soft_std": 0.049479253590106964, "reward_judge_quality_mean": 0.4437499940395355, "reward_judge_quality_std": 0.2084595113992691, "reward_total_composite_mean": 0.8210926055908203, "reward_total_composite_std": 0.0651756301522255} {"timestamp_utc": "2026-04-13T00:24:37Z", "mode": "train", "global_step": 869, "epoch": 0.08729281767955802, "loss": -0.0409, "grad_norm": 22.466726303100586, "learning_rate": 7.36969696969697e-06, "num_tokens": 1584481.0, "completions/mean_length": 18.5, "completions/min_length": 14.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 18.5, "completions/min_terminated_length": 14.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9868841767311096, "rewards/meter/std": 0.006352302152663469, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.946794867515564, "rewards/repeat_soft/std": 0.029547329992055893, "rewards/judge_quality/mean": 0.3725000023841858, "rewards/judge_quality/std": 0.11055056750774384, "rewards/total_composite/mean": 0.7102543115615845, "rewards/total_composite/std": 0.28757452964782715, "reward": 0.7102543115615845, "reward_std": 0.28757449984550476, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2004014402627945, "sampling/sampling_logp_difference/max": 1.5465419292449951, "sampling/importance_sampling_ratio/min": 0.21298320591449738, "sampling/importance_sampling_ratio/mean": 0.9888526797294617, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.118531495332718, "clip_ratio/low_mean": 0.008333333767950535, "clip_ratio/low_min": 0.008333333767950535, "clip_ratio/high_mean": 0.16878210147842765, "clip_ratio/high_max": 0.16878210147842765, "clip_ratio/region_mean": 0.17711543524637818, "reward_total_mean": 0.7102543115615845, "reward_meter_mean": 0.9868841767311096, "reward_meter_std": 0.006352302152663469, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.946794867515564, "reward_repeat_soft_std": 0.029547329992055893, "reward_judge_quality_mean": 0.3725000023841858, "reward_judge_quality_std": 0.11055056750774384, "reward_total_composite_mean": 0.7102543115615845, "reward_total_composite_std": 0.28757452964782715} {"timestamp_utc": "2026-04-13T00:24:44Z", "mode": "train", "global_step": 870, "epoch": 0.0873932697137117, "loss": 0.0189, "grad_norm": 16.99700164794922, "learning_rate": 7.3666666666666676e-06, "num_tokens": 1586151.0, "completions/mean_length": 34.75, "completions/min_length": 29.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.75, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.7278303503990173, "rewards/meter/std": 0.3372341990470886, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9471414089202881, "rewards/repeat_soft/std": 0.06687674671411514, "rewards/judge_quality/mean": 0.38374999165534973, "rewards/judge_quality/std": 0.1151319071650505, "rewards/total_composite/mean": 0.687362790107727, "rewards/total_composite/std": 0.17005886137485504, "reward": 0.687362790107727, "reward_std": 0.17005887627601624, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1950860619544983, "sampling/sampling_logp_difference/max": 1.6317987442016602, "sampling/importance_sampling_ratio/min": 0.1955774575471878, "sampling/importance_sampling_ratio/mean": 1.015047550201416, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6443248316645622, "clip_ratio/low_mean": 0.03623949736356735, "clip_ratio/low_min": 0.03623949736356735, "clip_ratio/high_mean": 0.11381876189261675, "clip_ratio/high_max": 0.11381876189261675, "clip_ratio/region_mean": 0.1500582592561841, "reward_total_mean": 0.687362790107727, "reward_meter_mean": 0.7278303503990173, "reward_meter_std": 0.3372341990470886, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9471414089202881, "reward_repeat_soft_std": 0.06687674671411514, "reward_judge_quality_mean": 0.38374999165534973, "reward_judge_quality_std": 0.1151319071650505, "reward_total_composite_mean": 0.687362790107727, "reward_total_composite_std": 0.17005886137485504} {"timestamp_utc": "2026-04-13T00:24:51Z", "mode": "train", "global_step": 871, "epoch": 0.08749372174786539, "loss": 0.1128, "grad_norm": 12.083643913269043, "learning_rate": 7.363636363636364e-06, "num_tokens": 1588346.0, "completions/mean_length": 75.375, "completions/min_length": 65.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.375, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.9300662875175476, "rewards/meter/std": 0.1772974133491516, "rewards/count_adherence/mean": 0.7749999761581421, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6679455041885376, "rewards/repeat_soft/std": 0.2263449728488922, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.12351980805397034, "rewards/total_composite/mean": 0.6945743560791016, "rewards/total_composite/std": 0.07519463449716568, "reward": 0.6945743560791016, "reward_std": 0.07519463449716568, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11953200399875641, "sampling/sampling_logp_difference/max": 1.6212282180786133, "sampling/importance_sampling_ratio/min": 0.1976557821035385, "sampling/importance_sampling_ratio/mean": 1.0150429010391235, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7786599360406399, "clip_ratio/low_mean": 0.02792270784266293, "clip_ratio/low_min": 0.02792270784266293, "clip_ratio/high_mean": 0.07752938289195299, "clip_ratio/high_max": 0.07752938289195299, "clip_ratio/region_mean": 0.10545209073461592, "reward_total_mean": 0.6945743560791016, "reward_meter_mean": 0.9300662875175476, "reward_meter_std": 0.1772974133491516, "reward_count_adherence_mean": 0.7749999761581421, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6679455041885376, "reward_repeat_soft_std": 0.2263449728488922, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.12351980805397034, "reward_total_composite_mean": 0.6945743560791016, "reward_total_composite_std": 0.07519463449716568} {"timestamp_utc": "2026-04-13T00:24:58Z", "mode": "train", "global_step": 872, "epoch": 0.08759417378201909, "loss": 0.0002, "grad_norm": 17.78493881225586, "learning_rate": 7.360606060606061e-06, "num_tokens": 1589910.0, "completions/mean_length": 34.5, "completions/min_length": 32.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.5, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.6330238580703735, "rewards/meter/std": 0.39083853363990784, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9839086532592773, "rewards/repeat_soft/std": 0.015805305913090706, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.10392305999994278, "rewards/total_composite/mean": 0.6727516055107117, "rewards/total_composite/std": 0.18868274986743927, "reward": 0.6727516055107117, "reward_std": 0.18868274986743927, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14930033683776855, "sampling/sampling_logp_difference/max": 1.811361312866211, "sampling/importance_sampling_ratio/min": 0.16343151032924652, "sampling/importance_sampling_ratio/mean": 1.000571846961975, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9098971337080002, "clip_ratio/low_mean": 0.04117627255618572, "clip_ratio/low_min": 0.04117627255618572, "clip_ratio/high_mean": 0.07740410882979631, "clip_ratio/high_max": 0.07740410882979631, "clip_ratio/region_mean": 0.11858038138598204, "reward_total_mean": 0.6727516055107117, "reward_meter_mean": 0.6330238580703735, "reward_meter_std": 0.39083853363990784, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9839086532592773, "reward_repeat_soft_std": 0.015805305913090706, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.10392305999994278, "reward_total_composite_mean": 0.6727516055107117, "reward_total_composite_std": 0.18868274986743927} {"timestamp_utc": "2026-04-13T00:25:05Z", "mode": "train", "global_step": 873, "epoch": 0.08769462581617278, "loss": -0.0381, "grad_norm": 10.22836685180664, "learning_rate": 7.357575757575758e-06, "num_tokens": 1591938.0, "completions/mean_length": 64.5, "completions/min_length": 55.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.5, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.990830659866333, "rewards/meter/std": 0.002650085836648941, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7830244898796082, "rewards/repeat_soft/std": 0.16684503853321075, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.22414520382881165, "rewards/total_composite/mean": 0.6681531667709351, "rewards/total_composite/std": 0.2796800434589386, "reward": 0.6681531667709351, "reward_std": 0.2796800136566162, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13292859494686127, "sampling/sampling_logp_difference/max": 1.2428040504455566, "sampling/importance_sampling_ratio/min": 0.28857389092445374, "sampling/importance_sampling_ratio/mean": 1.005682110786438, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9559882208704948, "clip_ratio/low_mean": 0.03427318320609629, "clip_ratio/low_min": 0.03427318320609629, "clip_ratio/high_mean": 0.08492259308695793, "clip_ratio/high_max": 0.08492259308695793, "clip_ratio/region_mean": 0.11919577629305422, "reward_total_mean": 0.6681531667709351, "reward_meter_mean": 0.990830659866333, "reward_meter_std": 0.002650085836648941, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7830244898796082, "reward_repeat_soft_std": 0.16684503853321075, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.22414520382881165, "reward_total_composite_mean": 0.6681531667709351, "reward_total_composite_std": 0.2796800434589386} {"timestamp_utc": "2026-04-13T00:25:12Z", "mode": "train", "global_step": 874, "epoch": 0.08779507785032648, "loss": 0.0091, "grad_norm": 16.014047622680664, "learning_rate": 7.354545454545456e-06, "num_tokens": 1593807.0, "completions/mean_length": 50.625, "completions/min_length": 39.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.625, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.7332473993301392, "rewards/meter/std": 0.27142488956451416, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9182817339897156, "rewards/repeat_soft/std": 0.15399900078773499, "rewards/judge_quality/mean": 0.6825000047683716, "rewards/judge_quality/std": 0.23260945081710815, "rewards/total_composite/mean": 0.7390395402908325, "rewards/total_composite/std": 0.10508356243371964, "reward": 0.7390395402908325, "reward_std": 0.10508355498313904, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14347051084041595, "sampling/sampling_logp_difference/max": 1.6096352338790894, "sampling/importance_sampling_ratio/min": 0.19996054470539093, "sampling/importance_sampling_ratio/mean": 0.9927876591682434, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8102206140756607, "clip_ratio/low_mean": 0.0520204147323966, "clip_ratio/low_min": 0.0520204147323966, "clip_ratio/high_mean": 0.08579732896760106, "clip_ratio/high_max": 0.08579732896760106, "clip_ratio/region_mean": 0.13781774369999766, "reward_total_mean": 0.7390395402908325, "reward_meter_mean": 0.7332473993301392, "reward_meter_std": 0.27142488956451416, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9182817339897156, "reward_repeat_soft_std": 0.15399900078773499, "reward_judge_quality_mean": 0.6825000047683716, "reward_judge_quality_std": 0.23260945081710815, "reward_total_composite_mean": 0.7390395402908325, "reward_total_composite_std": 0.10508356243371964} {"timestamp_utc": "2026-04-13T00:25:19Z", "mode": "train", "global_step": 875, "epoch": 0.08789552988448016, "loss": 0.0575, "grad_norm": 12.473429679870605, "learning_rate": 7.351515151515151e-06, "num_tokens": 1595468.0, "completions/mean_length": 39.625, "completions/min_length": 36.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.625, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.9891446828842163, "rewards/meter/std": 0.005651688668876886, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9578578472137451, "rewards/repeat_soft/std": 0.08193717151880264, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.8379008769989014, "rewards/total_composite/std": 0.05398271977901459, "reward": 0.8379008769989014, "reward_std": 0.05398271977901459, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17842237651348114, "sampling/sampling_logp_difference/max": 1.7161407470703125, "sampling/importance_sampling_ratio/min": 0.17975856363773346, "sampling/importance_sampling_ratio/mean": 1.0236784219741821, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2785840332508087, "clip_ratio/low_mean": 0.13195270020514727, "clip_ratio/low_min": 0.13195270020514727, "clip_ratio/high_mean": 0.02083333395421505, "clip_ratio/high_max": 0.02083333395421505, "clip_ratio/region_mean": 0.15278603415936232, "reward_total_mean": 0.8379008769989014, "reward_meter_mean": 0.9891446828842163, "reward_meter_std": 0.005651688668876886, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9578578472137451, "reward_repeat_soft_std": 0.08193717151880264, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.8379008769989014, "reward_total_composite_std": 0.05398271977901459} {"timestamp_utc": "2026-04-13T00:25:25Z", "mode": "train", "global_step": 876, "epoch": 0.08799598191863385, "loss": -0.0329, "grad_norm": 13.042718887329102, "learning_rate": 7.348484848484849e-06, "num_tokens": 1597206.0, "completions/mean_length": 34.25, "completions/min_length": 27.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.25, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.6461721658706665, "rewards/meter/std": 0.3770108222961426, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9856493473052979, "rewards/repeat_soft/std": 0.01778586208820343, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.25587037205696106, "rewards/total_composite/mean": 0.6225231885910034, "rewards/total_composite/std": 0.3225468397140503, "reward": 0.6225231885910034, "reward_std": 0.3225468397140503, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16878846287727356, "sampling/sampling_logp_difference/max": 1.6650335788726807, "sampling/importance_sampling_ratio/min": 0.189184308052063, "sampling/importance_sampling_ratio/mean": 0.9992000460624695, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1607022434473038, "clip_ratio/low_mean": 0.08212019596248865, "clip_ratio/low_min": 0.08212019596248865, "clip_ratio/high_mean": 0.10303995572030544, "clip_ratio/high_max": 0.10303995572030544, "clip_ratio/region_mean": 0.1851601516827941, "reward_total_mean": 0.6225231885910034, "reward_meter_mean": 0.6461721658706665, "reward_meter_std": 0.3770108222961426, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9856493473052979, "reward_repeat_soft_std": 0.01778586208820343, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.25587037205696106, "reward_total_composite_mean": 0.6225231885910034, "reward_total_composite_std": 0.3225468397140503} {"timestamp_utc": "2026-04-13T00:25:31Z", "mode": "train", "global_step": 877, "epoch": 0.08809643395278755, "loss": -0.0389, "grad_norm": 22.923341751098633, "learning_rate": 7.345454545454546e-06, "num_tokens": 1598509.0, "completions/mean_length": 23.875, "completions/min_length": 20.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.875, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9701720476150513, "rewards/meter/std": 0.02838454768061638, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9671835899353027, "rewards/repeat_soft/std": 0.02746698446571827, "rewards/judge_quality/mean": 0.8575000166893005, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.9405457973480225, "rewards/total_composite/std": 0.05239082872867584, "reward": 0.9405457973480225, "reward_std": 0.05239083990454674, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10772784054279327, "sampling/sampling_logp_difference/max": 5.0852789878845215, "sampling/importance_sampling_ratio/min": 0.00618716049939394, "sampling/importance_sampling_ratio/mean": 1.0057138204574585, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4652152396738529, "clip_ratio/low_mean": 0.028409091755747795, "clip_ratio/low_min": 0.028409091755747795, "clip_ratio/high_mean": 0.06643782602623105, "clip_ratio/high_max": 0.06643782602623105, "clip_ratio/region_mean": 0.09484691778197885, "reward_total_mean": 0.9405457973480225, "reward_meter_mean": 0.9701720476150513, "reward_meter_std": 0.02838454768061638, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9671835899353027, "reward_repeat_soft_std": 0.02746698446571827, "reward_judge_quality_mean": 0.8575000166893005, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.9405457973480225, "reward_total_composite_std": 0.05239082872867584} {"timestamp_utc": "2026-04-13T00:25:37Z", "mode": "train", "global_step": 878, "epoch": 0.08819688598694124, "loss": 0.0203, "grad_norm": 22.66485023498535, "learning_rate": 7.342424242424243e-06, "num_tokens": 1600081.0, "completions/mean_length": 29.5, "completions/min_length": 29.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.5, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.7032918334007263, "rewards/meter/std": 0.4175125062465668, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9532322883605957, "rewards/repeat_soft/std": 0.028042495250701904, "rewards/judge_quality/mean": 0.5900000333786011, "rewards/judge_quality/std": 0.27994900941848755, "rewards/total_composite/mean": 0.7388045787811279, "rewards/total_composite/std": 0.23565727472305298, "reward": 0.7388045787811279, "reward_std": 0.23565725982189178, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16955168545246124, "sampling/sampling_logp_difference/max": 2.1050338745117188, "sampling/importance_sampling_ratio/min": 0.12184154242277145, "sampling/importance_sampling_ratio/mean": 0.9927035570144653, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0104823261499405, "clip_ratio/low_mean": 0.0337643688544631, "clip_ratio/low_min": 0.0337643688544631, "clip_ratio/high_mean": 0.08850574912503362, "clip_ratio/high_max": 0.08850574912503362, "clip_ratio/region_mean": 0.12227011797949672, "reward_total_mean": 0.7388045787811279, "reward_meter_mean": 0.7032918334007263, "reward_meter_std": 0.4175125062465668, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9532322883605957, "reward_repeat_soft_std": 0.028042495250701904, "reward_judge_quality_mean": 0.5900000333786011, "reward_judge_quality_std": 0.27994900941848755, "reward_total_composite_mean": 0.7388045787811279, "reward_total_composite_std": 0.23565727472305298} {"timestamp_utc": "2026-04-13T00:25:44Z", "mode": "train", "global_step": 879, "epoch": 0.08829733802109492, "loss": 0.0584, "grad_norm": 26.22050666809082, "learning_rate": 7.3393939393939395e-06, "num_tokens": 1601419.0, "completions/mean_length": 25.25, "completions/min_length": 21.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.25, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.5514183044433594, "rewards/meter/std": 0.41531261801719666, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9963815212249756, "rewards/repeat_soft/std": 0.005040032789111137, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.6436513662338257, "rewards/total_composite/std": 0.21222326159477234, "reward": 0.6436513662338257, "reward_std": 0.21222324669361115, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1812986582517624, "sampling/sampling_logp_difference/max": 1.3460869789123535, "sampling/importance_sampling_ratio/min": 0.26275065541267395, "sampling/importance_sampling_ratio/mean": 1.0127619504928589, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.912125252187252, "clip_ratio/low_mean": 0.07158433133736253, "clip_ratio/low_min": 0.07158433133736253, "clip_ratio/high_mean": 0.09686756134033203, "clip_ratio/high_max": 0.09686756134033203, "clip_ratio/region_mean": 0.16845189267769456, "reward_total_mean": 0.6436513662338257, "reward_meter_mean": 0.5514183044433594, "reward_meter_std": 0.41531261801719666, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9963815212249756, "reward_repeat_soft_std": 0.005040032789111137, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.6436513662338257, "reward_total_composite_std": 0.21222326159477234} {"timestamp_utc": "2026-04-13T00:25:51Z", "mode": "train", "global_step": 880, "epoch": 0.08839779005524862, "loss": 0.0321, "grad_norm": 11.65561294555664, "learning_rate": 7.336363636363637e-06, "num_tokens": 1603063.0, "completions/mean_length": 40.5, "completions/min_length": 36.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.8762719631195068, "rewards/meter/std": 0.2661809027194977, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9715333580970764, "rewards/repeat_soft/std": 0.029223943129181862, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.6597867608070374, "rewards/total_composite/std": 0.28173133730888367, "reward": 0.6597867608070374, "reward_std": 0.28173133730888367, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15395426750183105, "sampling/sampling_logp_difference/max": 2.097630262374878, "sampling/importance_sampling_ratio/min": 0.12274695932865143, "sampling/importance_sampling_ratio/mean": 1.0011086463928223, "sampling/importance_sampling_ratio/max": 1.7908406257629395, "entropy": 1.1161795482039452, "clip_ratio/low_mean": 0.04452264495193958, "clip_ratio/low_min": 0.04452264495193958, "clip_ratio/high_mean": 0.10535806510597467, "clip_ratio/high_max": 0.10535806510597467, "clip_ratio/region_mean": 0.14988071005791426, "reward_total_mean": 0.6597867608070374, "reward_meter_mean": 0.8762719631195068, "reward_meter_std": 0.2661809027194977, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9715333580970764, "reward_repeat_soft_std": 0.029223943129181862, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.6597867608070374, "reward_total_composite_std": 0.28173133730888367} {"timestamp_utc": "2026-04-13T00:25:58Z", "mode": "train", "global_step": 881, "epoch": 0.08849824208940231, "loss": -0.0842, "grad_norm": 10.013266563415527, "learning_rate": 7.333333333333333e-06, "num_tokens": 1605417.0, "completions/mean_length": 66.25, "completions/min_length": 50.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.25, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9892548322677612, "rewards/meter/std": 0.010539655573666096, "rewards/count_adherence/mean": 0.65625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.5931967496871948, "rewards/repeat_soft/std": 0.25104108452796936, "rewards/judge_quality/mean": 0.23375000059604645, "rewards/judge_quality/std": 0.09006941318511963, "rewards/total_composite/mean": 0.6730468273162842, "rewards/total_composite/std": 0.06074142828583717, "reward": 0.6730468273162842, "reward_std": 0.06074143201112747, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12164386361837387, "sampling/sampling_logp_difference/max": 2.838268518447876, "sampling/importance_sampling_ratio/min": 0.058526914566755295, "sampling/importance_sampling_ratio/mean": 1.017610788345337, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7417355179786682, "clip_ratio/low_mean": 0.04895611107349396, "clip_ratio/low_min": 0.04895611107349396, "clip_ratio/high_mean": 0.06027026940137148, "clip_ratio/high_max": 0.06027026940137148, "clip_ratio/region_mean": 0.10922638047486544, "reward_total_mean": 0.6730468273162842, "reward_meter_mean": 0.9892548322677612, "reward_meter_std": 0.010539655573666096, "reward_count_adherence_mean": 0.65625, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.5931967496871948, "reward_repeat_soft_std": 0.25104108452796936, "reward_judge_quality_mean": 0.23375000059604645, "reward_judge_quality_std": 0.09006941318511963, "reward_total_composite_mean": 0.6730468273162842, "reward_total_composite_std": 0.06074142828583717} {"timestamp_utc": "2026-04-13T00:26:05Z", "mode": "train", "global_step": 882, "epoch": 0.088598694123556, "loss": 0.0759, "grad_norm": 13.193831443786621, "learning_rate": 7.330303030303031e-06, "num_tokens": 1607374.0, "completions/mean_length": 63.625, "completions/min_length": 52.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.625, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.8855563402175903, "rewards/meter/std": 0.183639258146286, "rewards/count_adherence/mean": 0.7749999761581421, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9259289503097534, "rewards/repeat_soft/std": 0.09029866009950638, "rewards/judge_quality/mean": 0.6074999570846558, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.7895932197570801, "rewards/total_composite/std": 0.09788843989372253, "reward": 0.7895932197570801, "reward_std": 0.09788843989372253, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15427488088607788, "sampling/sampling_logp_difference/max": 2.63920259475708, "sampling/importance_sampling_ratio/min": 0.07141819596290588, "sampling/importance_sampling_ratio/mean": 1.0151187181472778, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1599371805787086, "clip_ratio/low_mean": 0.04181226436048746, "clip_ratio/low_min": 0.04181226436048746, "clip_ratio/high_mean": 0.12963983044028282, "clip_ratio/high_max": 0.12963983044028282, "clip_ratio/region_mean": 0.17145209480077028, "reward_total_mean": 0.7895932197570801, "reward_meter_mean": 0.8855563402175903, "reward_meter_std": 0.183639258146286, "reward_count_adherence_mean": 0.7749999761581421, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9259289503097534, "reward_repeat_soft_std": 0.09029866009950638, "reward_judge_quality_mean": 0.6074999570846558, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.7895932197570801, "reward_total_composite_std": 0.09788843989372253} {"timestamp_utc": "2026-04-13T00:26:12Z", "mode": "train", "global_step": 883, "epoch": 0.0886991461577097, "loss": 0.0119, "grad_norm": 19.001028060913086, "learning_rate": 7.3272727272727285e-06, "num_tokens": 1609018.0, "completions/mean_length": 33.5, "completions/min_length": 32.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.5, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.8100236058235168, "rewards/meter/std": 0.3489779531955719, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9487527012825012, "rewards/repeat_soft/std": 0.08687741309404373, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.7398859262466431, "rewards/total_composite/std": 0.15776343643665314, "reward": 0.7398859262466431, "reward_std": 0.15776345133781433, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15416283905506134, "sampling/sampling_logp_difference/max": 1.6380259990692139, "sampling/importance_sampling_ratio/min": 0.19436334073543549, "sampling/importance_sampling_ratio/mean": 1.0556590557098389, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9965294152498245, "clip_ratio/low_mean": 0.05078125, "clip_ratio/low_min": 0.05078125, "clip_ratio/high_mean": 0.0978910755366087, "clip_ratio/high_max": 0.0978910755366087, "clip_ratio/region_mean": 0.1486723255366087, "reward_total_mean": 0.7398859262466431, "reward_meter_mean": 0.8100236058235168, "reward_meter_std": 0.3489779531955719, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9487527012825012, "reward_repeat_soft_std": 0.08687741309404373, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.7398859262466431, "reward_total_composite_std": 0.15776343643665314} {"timestamp_utc": "2026-04-13T00:26:19Z", "mode": "train", "global_step": 884, "epoch": 0.08879959819186338, "loss": 0.0616, "grad_norm": 13.052233695983887, "learning_rate": 7.324242424242425e-06, "num_tokens": 1611269.0, "completions/mean_length": 80.375, "completions/min_length": 64.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.375, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.8037976026535034, "rewards/meter/std": 0.28571754693984985, "rewards/count_adherence/mean": 0.7916666269302368, "rewards/count_adherence/std": 0.07715165615081787, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8614452481269836, "rewards/repeat_soft/std": 0.08262279629707336, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.6862284541130066, "rewards/total_composite/std": 0.12538328766822815, "reward": 0.6862284541130066, "reward_std": 0.12538327276706696, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15380127727985382, "sampling/sampling_logp_difference/max": 1.6129302978515625, "sampling/importance_sampling_ratio/min": 0.19930273294448853, "sampling/importance_sampling_ratio/mean": 1.034277319908142, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1308182775974274, "clip_ratio/low_mean": 0.04921486508101225, "clip_ratio/low_min": 0.04921486508101225, "clip_ratio/high_mean": 0.09056300716474652, "clip_ratio/high_max": 0.09056300716474652, "clip_ratio/region_mean": 0.13977787224575877, "reward_total_mean": 0.6862284541130066, "reward_meter_mean": 0.8037976026535034, "reward_meter_std": 0.28571754693984985, "reward_count_adherence_mean": 0.7916666269302368, "reward_count_adherence_std": 0.07715165615081787, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8614452481269836, "reward_repeat_soft_std": 0.08262279629707336, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.6862284541130066, "reward_total_composite_std": 0.12538328766822815} {"timestamp_utc": "2026-04-13T00:26:31Z", "mode": "train", "global_step": 885, "epoch": 0.08890005022601707, "loss": -0.1315, "grad_norm": 4.292393207550049, "learning_rate": 7.321212121212122e-06, "num_tokens": 1613065.0, "completions/mean_length": 110.5, "completions/min_length": 46.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 53.142860412597656, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.4544620215892792, "rewards/meter/std": 0.4360155463218689, "rewards/count_adherence/mean": 0.71875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9635623693466187, "rewards/repeat_soft/std": 0.05342383310198784, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.4800010919570923, "rewards/total_composite/std": 0.26752084493637085, "reward": 0.4800010919570923, "reward_std": 0.26752084493637085, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1747671365737915, "sampling/sampling_logp_difference/max": 1.4432792663574219, "sampling/importance_sampling_ratio/min": 0.23615208268165588, "sampling/importance_sampling_ratio/mean": 1.0039938688278198, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0475654229521751, "clip_ratio/low_mean": 0.06440483778715134, "clip_ratio/low_min": 0.06440483778715134, "clip_ratio/high_mean": 0.06461008451879025, "clip_ratio/high_max": 0.06461008451879025, "clip_ratio/region_mean": 0.12901492230594158, "reward_total_mean": 0.4800010919570923, "reward_meter_mean": 0.4544620215892792, "reward_meter_std": 0.4360155463218689, "reward_count_adherence_mean": 0.71875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9635623693466187, "reward_repeat_soft_std": 0.05342383310198784, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.4800010919570923, "reward_total_composite_std": 0.26752084493637085} {"timestamp_utc": "2026-04-13T00:26:38Z", "mode": "train", "global_step": 886, "epoch": 0.08900050226017077, "loss": -0.0099, "grad_norm": 16.664186477661133, "learning_rate": 7.3181818181818186e-06, "num_tokens": 1614628.0, "completions/mean_length": 28.375, "completions/min_length": 22.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.375, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.6767488718032837, "rewards/meter/std": 0.4377126395702362, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9594928026199341, "rewards/repeat_soft/std": 0.008505655452609062, "rewards/judge_quality/mean": 0.5149999856948853, "rewards/judge_quality/std": 0.2676885426044464, "rewards/total_composite/mean": 0.7049863338470459, "rewards/total_composite/std": 0.24385768175125122, "reward": 0.7049863338470459, "reward_std": 0.24385768175125122, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16706444323062897, "sampling/sampling_logp_difference/max": 1.5080220699310303, "sampling/importance_sampling_ratio/min": 0.2213473618030548, "sampling/importance_sampling_ratio/mean": 0.9971925616264343, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3281905204057693, "clip_ratio/low_mean": 0.07014106772840023, "clip_ratio/low_min": 0.07014106772840023, "clip_ratio/high_mean": 0.09788352251052856, "clip_ratio/high_max": 0.09788352251052856, "clip_ratio/region_mean": 0.1680245902389288, "reward_total_mean": 0.7049863338470459, "reward_meter_mean": 0.6767488718032837, "reward_meter_std": 0.4377126395702362, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9594928026199341, "reward_repeat_soft_std": 0.008505655452609062, "reward_judge_quality_mean": 0.5149999856948853, "reward_judge_quality_std": 0.2676885426044464, "reward_total_composite_mean": 0.7049863338470459, "reward_total_composite_std": 0.24385768175125122} {"timestamp_utc": "2026-04-13T00:26:45Z", "mode": "train", "global_step": 887, "epoch": 0.08910095429432446, "loss": -0.102, "grad_norm": 23.377470016479492, "learning_rate": 7.315151515151516e-06, "num_tokens": 1615942.0, "completions/mean_length": 20.25, "completions/min_length": 13.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.25, "completions/min_terminated_length": 13.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.3882331848144531, "rewards/meter/std": 0.43626436591148376, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5525799989700317, "rewards/total_composite/std": 0.19883856177330017, "reward": 0.5525799989700317, "reward_std": 0.19883854687213898, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16492697596549988, "sampling/sampling_logp_difference/max": 1.1572601795196533, "sampling/importance_sampling_ratio/min": 0.3143462538719177, "sampling/importance_sampling_ratio/mean": 1.0429223775863647, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4359506219625473, "clip_ratio/low_mean": 0.09593539871275425, "clip_ratio/low_min": 0.09593539871275425, "clip_ratio/high_mean": 0.060101200826466084, "clip_ratio/high_max": 0.060101200826466084, "clip_ratio/region_mean": 0.15603659953922033, "reward_total_mean": 0.5525799989700317, "reward_meter_mean": 0.3882331848144531, "reward_meter_std": 0.43626436591148376, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5525799989700317, "reward_total_composite_std": 0.19883856177330017} {"timestamp_utc": "2026-04-13T00:26:57Z", "mode": "train", "global_step": 888, "epoch": 0.08920140632847816, "loss": -0.1411, "grad_norm": 3.5283093452453613, "learning_rate": 7.312121212121212e-06, "num_tokens": 1617735.0, "completions/mean_length": 116.125, "completions/min_length": 52.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 59.57143020629883, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.6846745610237122, "rewards/meter/std": 0.3434545397758484, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9496748447418213, "rewards/repeat_soft/std": 0.03277288377285004, "rewards/judge_quality/mean": 0.375, "rewards/judge_quality/std": 0.15212775766849518, "rewards/total_composite/mean": 0.5666366815567017, "rewards/total_composite/std": 0.2768646776676178, "reward": 0.5666366815567017, "reward_std": 0.2768646776676178, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15418553352355957, "sampling/sampling_logp_difference/max": 1.7946438789367676, "sampling/importance_sampling_ratio/min": 0.1661866307258606, "sampling/importance_sampling_ratio/mean": 1.0316224098205566, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.959318220615387, "clip_ratio/low_mean": 0.05048076994717121, "clip_ratio/low_min": 0.05048076994717121, "clip_ratio/high_mean": 0.09615987446159124, "clip_ratio/high_max": 0.09615987446159124, "clip_ratio/region_mean": 0.14664064440876245, "reward_total_mean": 0.5666366815567017, "reward_meter_mean": 0.6846745610237122, "reward_meter_std": 0.3434545397758484, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9496748447418213, "reward_repeat_soft_std": 0.03277288377285004, "reward_judge_quality_mean": 0.375, "reward_judge_quality_std": 0.15212775766849518, "reward_total_composite_mean": 0.5666366815567017, "reward_total_composite_std": 0.2768646776676178} {"timestamp_utc": "2026-04-13T00:27:04Z", "mode": "train", "global_step": 889, "epoch": 0.08930185836263184, "loss": 0.0479, "grad_norm": 17.075801849365234, "learning_rate": 7.30909090909091e-06, "num_tokens": 1619220.0, "completions/mean_length": 32.625, "completions/min_length": 23.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.625, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.25817200541496277, "rewards/meter/std": 0.3672030568122864, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.17251639068126678, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9895851612091064, "rewards/repeat_soft/std": 0.004242531023919582, "rewards/judge_quality/mean": 0.7487500309944153, "rewards/judge_quality/std": 0.21853001415729523, "rewards/total_composite/mean": 0.571010947227478, "rewards/total_composite/std": 0.12429411709308624, "reward": 0.571010947227478, "reward_std": 0.12429411709308624, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1067616194486618, "sampling/sampling_logp_difference/max": 1.0465302467346191, "sampling/importance_sampling_ratio/min": 0.413419246673584, "sampling/importance_sampling_ratio/mean": 1.0453273057937622, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7065118737518787, "clip_ratio/low_mean": 0.05780586553737521, "clip_ratio/low_min": 0.05780586553737521, "clip_ratio/high_mean": 0.038541666232049465, "clip_ratio/high_max": 0.038541666232049465, "clip_ratio/region_mean": 0.09634753176942468, "reward_total_mean": 0.571010947227478, "reward_meter_mean": 0.25817200541496277, "reward_meter_std": 0.3672030568122864, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.17251639068126678, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9895851612091064, "reward_repeat_soft_std": 0.004242531023919582, "reward_judge_quality_mean": 0.7487500309944153, "reward_judge_quality_std": 0.21853001415729523, "reward_total_composite_mean": 0.571010947227478, "reward_total_composite_std": 0.12429411709308624} {"timestamp_utc": "2026-04-13T00:27:12Z", "mode": "train", "global_step": 890, "epoch": 0.08940231039678553, "loss": 0.0953, "grad_norm": 14.652690887451172, "learning_rate": 7.306060606060607e-06, "num_tokens": 1620696.0, "completions/mean_length": 34.5, "completions/min_length": 32.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.5, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.8639898896217346, "rewards/meter/std": 0.16379791498184204, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9839224815368652, "rewards/repeat_soft/std": 0.02250523865222931, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.7601877450942993, "rewards/total_composite/std": 0.08510161191225052, "reward": 0.7601877450942993, "reward_std": 0.08510160446166992, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17874585092067719, "sampling/sampling_logp_difference/max": 3.1429879665374756, "sampling/importance_sampling_ratio/min": 0.13888172805309296, "sampling/importance_sampling_ratio/mean": 1.030044436454773, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1074253916740417, "clip_ratio/low_mean": 0.04969113413244486, "clip_ratio/low_min": 0.04969113413244486, "clip_ratio/high_mean": 0.09989177715033293, "clip_ratio/high_max": 0.09989177715033293, "clip_ratio/region_mean": 0.1495829112827778, "reward_total_mean": 0.7601877450942993, "reward_meter_mean": 0.8639898896217346, "reward_meter_std": 0.16379791498184204, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9839224815368652, "reward_repeat_soft_std": 0.02250523865222931, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.7601877450942993, "reward_total_composite_std": 0.08510161191225052} {"timestamp_utc": "2026-04-13T00:27:18Z", "mode": "train", "global_step": 891, "epoch": 0.08950276243093923, "loss": 0.0219, "grad_norm": 27.928890228271484, "learning_rate": 7.303030303030304e-06, "num_tokens": 1622101.0, "completions/mean_length": 23.625, "completions/min_length": 20.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.625, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.2756759524345398, "rewards/meter/std": 0.4389852285385132, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9991971850395203, "rewards/repeat_soft/std": 0.001237936899997294, "rewards/judge_quality/mean": 0.690000057220459, "rewards/judge_quality/std": 0.22884805500507355, "rewards/total_composite/mean": 0.5809738636016846, "rewards/total_composite/std": 0.20783966779708862, "reward": 0.5809738636016846, "reward_std": 0.20783966779708862, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2005946934223175, "sampling/sampling_logp_difference/max": 1.6689624786376953, "sampling/importance_sampling_ratio/min": 0.18844248354434967, "sampling/importance_sampling_ratio/mean": 1.0314934253692627, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8127978891134262, "clip_ratio/low_mean": 0.11869451031088829, "clip_ratio/low_min": 0.11869451031088829, "clip_ratio/high_mean": 0.06384615413844585, "clip_ratio/high_max": 0.06384615413844585, "clip_ratio/region_mean": 0.18254066444933414, "reward_total_mean": 0.5809738636016846, "reward_meter_mean": 0.2756759524345398, "reward_meter_std": 0.4389852285385132, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9991971850395203, "reward_repeat_soft_std": 0.001237936899997294, "reward_judge_quality_mean": 0.690000057220459, "reward_judge_quality_std": 0.22884805500507355, "reward_total_composite_mean": 0.5809738636016846, "reward_total_composite_std": 0.20783966779708862} {"timestamp_utc": "2026-04-13T00:27:26Z", "mode": "train", "global_step": 892, "epoch": 0.08960321446509292, "loss": 0.0159, "grad_norm": 9.926685333251953, "learning_rate": 7.3e-06, "num_tokens": 1623941.0, "completions/mean_length": 55.0, "completions/min_length": 50.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.0, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9866457581520081, "rewards/meter/std": 0.010019187815487385, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9825280904769897, "rewards/repeat_soft/std": 0.018980614840984344, "rewards/judge_quality/mean": 0.48375001549720764, "rewards/judge_quality/std": 0.18965667486190796, "rewards/total_composite/mean": 0.8373684287071228, "rewards/total_composite/std": 0.05813080444931984, "reward": 0.8373684287071228, "reward_std": 0.05813080072402954, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1612224131822586, "sampling/sampling_logp_difference/max": 1.3584221601486206, "sampling/importance_sampling_ratio/min": 0.2570660710334778, "sampling/importance_sampling_ratio/mean": 0.9949221611022949, "sampling/importance_sampling_ratio/max": 1.8493988513946533, "entropy": 1.1537452712655067, "clip_ratio/low_mean": 0.1084195519797504, "clip_ratio/low_min": 0.1084195519797504, "clip_ratio/high_mean": 0.023148147389292717, "clip_ratio/high_max": 0.023148147389292717, "clip_ratio/region_mean": 0.1315676993690431, "reward_total_mean": 0.8373684287071228, "reward_meter_mean": 0.9866457581520081, "reward_meter_std": 0.010019187815487385, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9825280904769897, "reward_repeat_soft_std": 0.018980614840984344, "reward_judge_quality_mean": 0.48375001549720764, "reward_judge_quality_std": 0.18965667486190796, "reward_total_composite_mean": 0.8373684287071228, "reward_total_composite_std": 0.05813080444931984} {"timestamp_utc": "2026-04-13T00:27:34Z", "mode": "train", "global_step": 893, "epoch": 0.0897036664992466, "loss": 0.0579, "grad_norm": 19.848304748535156, "learning_rate": 7.296969696969698e-06, "num_tokens": 1625386.0, "completions/mean_length": 33.625, "completions/min_length": 31.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.625, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.635492742061615, "rewards/meter/std": 0.4111209809780121, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9901236295700073, "rewards/repeat_soft/std": 0.004412362817674875, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.6831091046333313, "rewards/total_composite/std": 0.20873039960861206, "reward": 0.6831091046333313, "reward_std": 0.20873038470745087, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1866736263036728, "sampling/sampling_logp_difference/max": 2.378268241882324, "sampling/importance_sampling_ratio/min": 0.09271098673343658, "sampling/importance_sampling_ratio/mean": 0.9839540719985962, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1274626478552818, "clip_ratio/low_mean": 0.046935468912124634, "clip_ratio/low_min": 0.046935468912124634, "clip_ratio/high_mean": 0.10206653364002705, "clip_ratio/high_max": 0.10206653364002705, "clip_ratio/region_mean": 0.14900200255215168, "reward_total_mean": 0.6831091046333313, "reward_meter_mean": 0.635492742061615, "reward_meter_std": 0.4111209809780121, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9901236295700073, "reward_repeat_soft_std": 0.004412362817674875, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.6831091046333313, "reward_total_composite_std": 0.20873039960861206} {"timestamp_utc": "2026-04-13T00:27:40Z", "mode": "train", "global_step": 894, "epoch": 0.0898041185334003, "loss": 0.0694, "grad_norm": 22.051578521728516, "learning_rate": 7.293939393939394e-06, "num_tokens": 1626808.0, "completions/mean_length": 18.75, "completions/min_length": 15.0, "completions/max_length": 22.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 18.75, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 22.0, "rewards/meter/mean": 0.924480676651001, "rewards/meter/std": 0.10070353001356125, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9594928026199341, "rewards/repeat_soft/std": 0.008505655452609062, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.8817155957221985, "rewards/total_composite/std": 0.0907624140381813, "reward": 0.8817155957221985, "reward_std": 0.09076240658760071, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1523188054561615, "sampling/sampling_logp_difference/max": 1.5500173568725586, "sampling/importance_sampling_ratio/min": 0.21224430203437805, "sampling/importance_sampling_ratio/mean": 1.0123330354690552, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0270814970135689, "clip_ratio/low_mean": 0.07350713293999434, "clip_ratio/low_min": 0.07350713293999434, "clip_ratio/high_mean": 0.08272058935835958, "clip_ratio/high_max": 0.08272058935835958, "clip_ratio/region_mean": 0.1562277222983539, "reward_total_mean": 0.8817155957221985, "reward_meter_mean": 0.924480676651001, "reward_meter_std": 0.10070353001356125, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9594928026199341, "reward_repeat_soft_std": 0.008505655452609062, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.8817155957221985, "reward_total_composite_std": 0.0907624140381813} {"timestamp_utc": "2026-04-13T00:27:48Z", "mode": "train", "global_step": 895, "epoch": 0.08990457056755399, "loss": -0.048, "grad_norm": 11.083358764648438, "learning_rate": 7.290909090909092e-06, "num_tokens": 1628569.0, "completions/mean_length": 51.125, "completions/min_length": 42.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.125, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.8679274320602417, "rewards/meter/std": 0.3119673430919647, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9827500581741333, "rewards/repeat_soft/std": 0.023572038859128952, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.25150617957115173, "rewards/total_composite/mean": 0.7459672689437866, "rewards/total_composite/std": 0.17005707323551178, "reward": 0.7459672689437866, "reward_std": 0.17005707323551178, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.144117072224617, "sampling/sampling_logp_difference/max": 1.6663410663604736, "sampling/importance_sampling_ratio/min": 0.18893709778785706, "sampling/importance_sampling_ratio/mean": 1.0156607627868652, "sampling/importance_sampling_ratio/max": 1.9621096849441528, "entropy": 0.9416538849473, "clip_ratio/low_mean": 0.037698413245379925, "clip_ratio/low_min": 0.037698413245379925, "clip_ratio/high_mean": 0.08662695833481848, "clip_ratio/high_max": 0.08662695833481848, "clip_ratio/region_mean": 0.12432537158019841, "reward_total_mean": 0.7459672689437866, "reward_meter_mean": 0.8679274320602417, "reward_meter_std": 0.3119673430919647, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9827500581741333, "reward_repeat_soft_std": 0.023572038859128952, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.25150617957115173, "reward_total_composite_mean": 0.7459672689437866, "reward_total_composite_std": 0.17005707323551178} {"timestamp_utc": "2026-04-13T00:27:55Z", "mode": "train", "global_step": 896, "epoch": 0.09000502260170769, "loss": 0.0754, "grad_norm": 25.358684539794922, "learning_rate": 7.287878787878789e-06, "num_tokens": 1630162.0, "completions/mean_length": 28.125, "completions/min_length": 23.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.125, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.8497214913368225, "rewards/meter/std": 0.270576149225235, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9550833702087402, "rewards/repeat_soft/std": 0.05472878739237785, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7550079822540283, "rewards/total_composite/std": 0.12161900103092194, "reward": 0.7550079822540283, "reward_std": 0.12161900103092194, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14473728835582733, "sampling/sampling_logp_difference/max": 1.1751008033752441, "sampling/importance_sampling_ratio/min": 0.30878785252571106, "sampling/importance_sampling_ratio/mean": 1.0225324630737305, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7383250370621681, "clip_ratio/low_mean": 0.02083333395421505, "clip_ratio/low_min": 0.02083333395421505, "clip_ratio/high_mean": 0.11575482878834009, "clip_ratio/high_max": 0.11575482878834009, "clip_ratio/region_mean": 0.13658816274255514, "reward_total_mean": 0.7550079822540283, "reward_meter_mean": 0.8497214913368225, "reward_meter_std": 0.270576149225235, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9550833702087402, "reward_repeat_soft_std": 0.05472878739237785, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7550079822540283, "reward_total_composite_std": 0.12161900103092194} {"timestamp_utc": "2026-04-13T00:28:02Z", "mode": "train", "global_step": 897, "epoch": 0.09010547463586138, "loss": 0.0373, "grad_norm": 16.86933708190918, "learning_rate": 7.284848484848486e-06, "num_tokens": 1631660.0, "completions/mean_length": 38.25, "completions/min_length": 35.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.25, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.5579719543457031, "rewards/meter/std": 0.3958151936531067, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9953708648681641, "rewards/repeat_soft/std": 0.005544096231460571, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404937028885, "rewards/total_composite/mean": 0.6674994230270386, "rewards/total_composite/std": 0.13414764404296875, "reward": 0.6674994230270386, "reward_std": 0.13414762914180756, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1514360010623932, "sampling/sampling_logp_difference/max": 0.9703068733215332, "sampling/importance_sampling_ratio/min": 0.37896671891212463, "sampling/importance_sampling_ratio/mean": 1.025296688079834, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0852379649877548, "clip_ratio/low_mean": 0.056597222574055195, "clip_ratio/low_min": 0.056597222574055195, "clip_ratio/high_mean": 0.06936479127034545, "clip_ratio/high_max": 0.06936479127034545, "clip_ratio/region_mean": 0.12596201384440064, "reward_total_mean": 0.6674994230270386, "reward_meter_mean": 0.5579719543457031, "reward_meter_std": 0.3958151936531067, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9953708648681641, "reward_repeat_soft_std": 0.005544096231460571, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404937028885, "reward_total_composite_mean": 0.6674994230270386, "reward_total_composite_std": 0.13414764404296875} {"timestamp_utc": "2026-04-13T00:28:14Z", "mode": "train", "global_step": 898, "epoch": 0.09020592667001506, "loss": -0.0488, "grad_norm": 3.3161540031433105, "learning_rate": 7.281818181818182e-06, "num_tokens": 1632945.0, "completions/mean_length": 78.625, "completions/min_length": 15.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 16.71428680419922, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 18.0, "rewards/meter/mean": 0.7407711744308472, "rewards/meter/std": 0.4550321102142334, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9594065546989441, "rewards/repeat_soft/std": 0.008749538101255894, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.1403057724237442, "rewards/total_composite/mean": 0.6471189856529236, "rewards/total_composite/std": 0.2998858392238617, "reward": 0.6471189856529236, "reward_std": 0.2998858094215393, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20817448198795319, "sampling/sampling_logp_difference/max": 1.8523635864257812, "sampling/importance_sampling_ratio/min": 0.15686596930027008, "sampling/importance_sampling_ratio/mean": 1.0333366394042969, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9066524431109428, "clip_ratio/low_mean": 0.0069444444961845875, "clip_ratio/low_min": 0.0069444444961845875, "clip_ratio/high_mean": 0.1468954267911613, "clip_ratio/high_max": 0.1468954267911613, "clip_ratio/region_mean": 0.1538398712873459, "reward_total_mean": 0.6471189856529236, "reward_meter_mean": 0.7407711744308472, "reward_meter_std": 0.4550321102142334, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9594065546989441, "reward_repeat_soft_std": 0.008749538101255894, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.1403057724237442, "reward_total_composite_mean": 0.6471189856529236, "reward_total_composite_std": 0.2998858392238617} {"timestamp_utc": "2026-04-13T00:28:21Z", "mode": "train", "global_step": 899, "epoch": 0.09030637870416876, "loss": -0.0375, "grad_norm": 16.460742950439453, "learning_rate": 7.2787878787878795e-06, "num_tokens": 1634537.0, "completions/mean_length": 39.0, "completions/min_length": 35.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.8601393699645996, "rewards/meter/std": 0.24162888526916504, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9289783239364624, "rewards/repeat_soft/std": 0.07654180377721786, "rewards/judge_quality/mean": 0.7775000333786011, "rewards/judge_quality/std": 0.21952873468399048, "rewards/total_composite/mean": 0.8632105588912964, "rewards/total_composite/std": 0.1044643297791481, "reward": 0.8632105588912964, "reward_std": 0.10446430742740631, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17651066184043884, "sampling/sampling_logp_difference/max": 1.6718378067016602, "sampling/importance_sampling_ratio/min": 0.18790142238140106, "sampling/importance_sampling_ratio/mean": 1.031165361404419, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0943117141723633, "clip_ratio/low_mean": 0.062162162736058235, "clip_ratio/low_min": 0.062162162736058235, "clip_ratio/high_mean": 0.09198241867125034, "clip_ratio/high_max": 0.09198241867125034, "clip_ratio/region_mean": 0.15414458140730858, "reward_total_mean": 0.8632105588912964, "reward_meter_mean": 0.8601393699645996, "reward_meter_std": 0.24162888526916504, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9289783239364624, "reward_repeat_soft_std": 0.07654180377721786, "reward_judge_quality_mean": 0.7775000333786011, "reward_judge_quality_std": 0.21952873468399048, "reward_total_composite_mean": 0.8632105588912964, "reward_total_composite_std": 0.1044643297791481} {"timestamp_utc": "2026-04-13T00:28:28Z", "mode": "train", "global_step": 900, "epoch": 0.09040683073832245, "loss": -0.0005, "grad_norm": 9.337118148803711, "learning_rate": 7.275757575757576e-06, "num_tokens": 1636481.0, "completions/mean_length": 55.0, "completions/min_length": 48.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.0, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.8311859369277954, "rewards/meter/std": 0.302761971950531, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9413772821426392, "rewards/repeat_soft/std": 0.0711512491106987, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.7254214286804199, "rewards/total_composite/std": 0.17054739594459534, "reward": 0.7254214286804199, "reward_std": 0.17054738104343414, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16154694557189941, "sampling/sampling_logp_difference/max": 2.748352289199829, "sampling/importance_sampling_ratio/min": 0.06403328478336334, "sampling/importance_sampling_ratio/mean": 1.0268787145614624, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1935032904148102, "clip_ratio/low_mean": 0.049694056622684, "clip_ratio/low_min": 0.049694056622684, "clip_ratio/high_mean": 0.10413110349327326, "clip_ratio/high_max": 0.10413110349327326, "clip_ratio/region_mean": 0.15382516011595726, "reward_total_mean": 0.7254214286804199, "reward_meter_mean": 0.8311859369277954, "reward_meter_std": 0.302761971950531, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9413772821426392, "reward_repeat_soft_std": 0.0711512491106987, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.7254214286804199, "reward_total_composite_std": 0.17054739594459534} {"timestamp_utc": "2026-04-13T00:29:24Z", "mode": "eval", "global_step": 900, "epoch": 0.09040683073832245, "eval_loss": NaN, "eval_runtime": 56.2653, "eval_samples_per_second": 1.422, "eval_steps_per_second": 0.178, "eval_num_tokens": 1636481.0, "eval_completions/mean_length": 62.8125, "eval_completions/min_length": 27.4, "eval_completions/max_length": 187.4, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 51.276786422729494, "eval_completions/min_terminated_length": 27.4, "eval_completions/max_terminated_length": 100.8, "eval_rewards/meter/mean": 0.7519857168197632, "eval_rewards/meter/std": 0.3122325256466866, "eval_rewards/count_adherence/mean": 0.8193750023841858, "eval_rewards/count_adherence/std": 0.1440672144293785, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.07071067690849304, "eval_rewards/repeat_soft/mean": 0.9528588235378266, "eval_rewards/repeat_soft/std": 0.07263523619621992, "eval_rewards/judge_quality/mean": 0.45212499499320985, "eval_rewards/judge_quality/std": 0.18833467960357667, "eval_rewards/total_composite/mean": 0.6872850716114044, "eval_rewards/total_composite/std": 0.17901606038212775, "eval_reward": 0.6872850716114044, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.09126125499606133, "eval_sampling/sampling_logp_difference/max": 1.0745189189910889, "eval_sampling/importance_sampling_ratio/min": 0.37743838876485825, "eval_sampling/importance_sampling_ratio/mean": 1.0206343054771423, "eval_sampling/importance_sampling_ratio/max": 1.5052820920944214, "eval_entropy": 1.1831181049346924, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6872850716114044, "eval_reward_meter_mean": 0.7519857168197632, "eval_reward_meter_std": 0.3122325256466866, "eval_reward_count_adherence_mean": 0.8193750023841858, "eval_reward_count_adherence_std": 0.1440672144293785, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.07071067690849304, "eval_reward_repeat_soft_mean": 0.9528588235378266, "eval_reward_repeat_soft_std": 0.07263523619621992, "eval_reward_judge_quality_mean": 0.45212499499320985, "eval_reward_judge_quality_std": 0.18833467960357667, "eval_reward_total_composite_mean": 0.6872850716114044, "eval_reward_total_composite_std": 0.17901606038212775} {"timestamp_utc": "2026-04-13T00:29:35Z", "mode": "train", "global_step": 901, "epoch": 0.09050728277247615, "loss": 0.0293, "grad_norm": 25.306673049926758, "learning_rate": 7.272727272727273e-06, "num_tokens": 1637988.0, "completions/mean_length": 28.375, "completions/min_length": 25.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.375, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.7070843577384949, "rewards/meter/std": 0.3851107358932495, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8223175406455994, "rewards/repeat_soft/std": 0.09804663807153702, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.751419723033905, "rewards/total_composite/std": 0.17962917685508728, "reward": 0.751419723033905, "reward_std": 0.1796291619539261, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14723101258277893, "sampling/sampling_logp_difference/max": 1.2161102294921875, "sampling/importance_sampling_ratio/min": 0.29638075828552246, "sampling/importance_sampling_ratio/mean": 1.0226225852966309, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0630191639065742, "clip_ratio/low_mean": 0.048954326659440994, "clip_ratio/low_min": 0.048954326659440994, "clip_ratio/high_mean": 0.09548599738627672, "clip_ratio/high_max": 0.09548599738627672, "clip_ratio/region_mean": 0.14444032404571772, "reward_total_mean": 0.751419723033905, "reward_meter_mean": 0.7070843577384949, "reward_meter_std": 0.3851107358932495, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8223175406455994, "reward_repeat_soft_std": 0.09804663807153702, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.751419723033905, "reward_total_composite_std": 0.17962917685508728} {"timestamp_utc": "2026-04-13T00:29:42Z", "mode": "train", "global_step": 902, "epoch": 0.09060773480662983, "loss": 0.0132, "grad_norm": 18.12690544128418, "learning_rate": 7.26969696969697e-06, "num_tokens": 1639468.0, "completions/mean_length": 28.0, "completions/min_length": 24.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.0, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.810047447681427, "rewards/meter/std": 0.28085213899612427, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9942691922187805, "rewards/repeat_soft/std": 0.006983870640397072, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.7131983041763306, "rewards/total_composite/std": 0.14376585185527802, "reward": 0.7131983041763306, "reward_std": 0.14376585185527802, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16083073616027832, "sampling/sampling_logp_difference/max": 1.6915661096572876, "sampling/importance_sampling_ratio/min": 0.184230774641037, "sampling/importance_sampling_ratio/mean": 1.0170209407806396, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9018732756376266, "clip_ratio/low_mean": 0.05563186854124069, "clip_ratio/low_min": 0.05563186854124069, "clip_ratio/high_mean": 0.12942449981346726, "clip_ratio/high_max": 0.12942449981346726, "clip_ratio/region_mean": 0.18505636835470796, "reward_total_mean": 0.7131983041763306, "reward_meter_mean": 0.810047447681427, "reward_meter_std": 0.28085213899612427, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9942691922187805, "reward_repeat_soft_std": 0.006983870640397072, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.7131983041763306, "reward_total_composite_std": 0.14376585185527802} {"timestamp_utc": "2026-04-13T00:29:50Z", "mode": "train", "global_step": 903, "epoch": 0.09070818684078352, "loss": 0.0413, "grad_norm": 28.1905460357666, "learning_rate": 7.266666666666668e-06, "num_tokens": 1641079.0, "completions/mean_length": 25.375, "completions/min_length": 21.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.375, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.6102509498596191, "rewards/meter/std": 0.38283562660217285, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9743432402610779, "rewards/repeat_soft/std": 0.034990549087524414, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.0843462198972702, "rewards/total_composite/mean": 0.6375472545623779, "rewards/total_composite/std": 0.16135767102241516, "reward": 0.6375472545623779, "reward_std": 0.16135767102241516, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17235241830348969, "sampling/sampling_logp_difference/max": 1.43924880027771, "sampling/importance_sampling_ratio/min": 0.2371058166027069, "sampling/importance_sampling_ratio/mean": 1.0144888162612915, "sampling/importance_sampling_ratio/max": 1.9171323776245117, "entropy": 1.0332181379199028, "clip_ratio/low_mean": 0.06158424960449338, "clip_ratio/low_min": 0.06158424960449338, "clip_ratio/high_mean": 0.07840988785028458, "clip_ratio/high_max": 0.07840988785028458, "clip_ratio/region_mean": 0.13999413745477796, "reward_total_mean": 0.6375472545623779, "reward_meter_mean": 0.6102509498596191, "reward_meter_std": 0.38283562660217285, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9743432402610779, "reward_repeat_soft_std": 0.034990549087524414, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.0843462198972702, "reward_total_composite_mean": 0.6375472545623779, "reward_total_composite_std": 0.16135767102241516} {"timestamp_utc": "2026-04-13T00:29:59Z", "mode": "train", "global_step": 904, "epoch": 0.09080863887493722, "loss": 0.0216, "grad_norm": 15.340963363647461, "learning_rate": 7.263636363636364e-06, "num_tokens": 1642704.0, "completions/mean_length": 33.125, "completions/min_length": 31.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.125, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.8512524366378784, "rewards/meter/std": 0.26025280356407166, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9663740396499634, "rewards/repeat_soft/std": 0.02770637720823288, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.745451033115387, "rewards/total_composite/std": 0.1531819999217987, "reward": 0.745451033115387, "reward_std": 0.1531819850206375, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20800749957561493, "sampling/sampling_logp_difference/max": 1.9941730499267578, "sampling/importance_sampling_ratio/min": 0.13612617552280426, "sampling/importance_sampling_ratio/mean": 1.0145882368087769, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3503071665763855, "clip_ratio/low_mean": 0.06545928120613098, "clip_ratio/low_min": 0.06545928120613098, "clip_ratio/high_mean": 0.11343049723654985, "clip_ratio/high_max": 0.11343049723654985, "clip_ratio/region_mean": 0.17888977844268084, "reward_total_mean": 0.745451033115387, "reward_meter_mean": 0.8512524366378784, "reward_meter_std": 0.26025280356407166, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9663740396499634, "reward_repeat_soft_std": 0.02770637720823288, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.745451033115387, "reward_total_composite_std": 0.1531819999217987} {"timestamp_utc": "2026-04-13T00:30:06Z", "mode": "train", "global_step": 905, "epoch": 0.09090909090909091, "loss": 0.0754, "grad_norm": 16.72256088256836, "learning_rate": 7.260606060606061e-06, "num_tokens": 1644120.0, "completions/mean_length": 26.0, "completions/min_length": 18.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.0, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.6860920190811157, "rewards/meter/std": 0.4055790901184082, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9671272039413452, "rewards/repeat_soft/std": 0.013087818399071693, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.683704137802124, "rewards/total_composite/std": 0.1928628534078598, "reward": 0.683704137802124, "reward_std": 0.1928628385066986, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14179162681102753, "sampling/sampling_logp_difference/max": 1.1578130722045898, "sampling/importance_sampling_ratio/min": 0.31417250633239746, "sampling/importance_sampling_ratio/mean": 1.0132758617401123, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9798845052719116, "clip_ratio/low_mean": 0.09955186396837234, "clip_ratio/low_min": 0.09955186396837234, "clip_ratio/high_mean": 0.06499100429937243, "clip_ratio/high_max": 0.06499100429937243, "clip_ratio/region_mean": 0.16454286826774478, "reward_total_mean": 0.683704137802124, "reward_meter_mean": 0.6860920190811157, "reward_meter_std": 0.4055790901184082, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9671272039413452, "reward_repeat_soft_std": 0.013087818399071693, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.683704137802124, "reward_total_composite_std": 0.1928628534078598} {"timestamp_utc": "2026-04-13T00:30:13Z", "mode": "train", "global_step": 906, "epoch": 0.0910095429432446, "loss": 0.0597, "grad_norm": 12.067218780517578, "learning_rate": 7.257575757575758e-06, "num_tokens": 1646279.0, "completions/mean_length": 63.875, "completions/min_length": 58.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.875, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.903562068939209, "rewards/meter/std": 0.2366390824317932, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9209731817245483, "rewards/repeat_soft/std": 0.09525564312934875, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.24833375215530396, "rewards/total_composite/mean": 0.770575225353241, "rewards/total_composite/std": 0.14046362042427063, "reward": 0.770575225353241, "reward_std": 0.14046362042427063, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13399745523929596, "sampling/sampling_logp_difference/max": 1.7312355041503906, "sampling/importance_sampling_ratio/min": 0.17706550657749176, "sampling/importance_sampling_ratio/mean": 1.014706015586853, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9985558539628983, "clip_ratio/low_mean": 0.03025928419083357, "clip_ratio/low_min": 0.03025928419083357, "clip_ratio/high_mean": 0.09078721236437559, "clip_ratio/high_max": 0.09078721236437559, "clip_ratio/region_mean": 0.12104649655520916, "reward_total_mean": 0.770575225353241, "reward_meter_mean": 0.903562068939209, "reward_meter_std": 0.2366390824317932, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9209731817245483, "reward_repeat_soft_std": 0.09525564312934875, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.24833375215530396, "reward_total_composite_mean": 0.770575225353241, "reward_total_composite_std": 0.14046362042427063} {"timestamp_utc": "2026-04-13T00:30:21Z", "mode": "train", "global_step": 907, "epoch": 0.09110999497739829, "loss": 0.0532, "grad_norm": 15.94900131225586, "learning_rate": 7.254545454545455e-06, "num_tokens": 1648019.0, "completions/mean_length": 49.5, "completions/min_length": 43.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.5, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9180644750595093, "rewards/meter/std": 0.1755811870098114, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9243527054786682, "rewards/repeat_soft/std": 0.06503698229789734, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7346892952919006, "rewards/total_composite/std": 0.0747465193271637, "reward": 0.7346892952919006, "reward_std": 0.0747465193271637, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16690056025981903, "sampling/sampling_logp_difference/max": 1.5347201824188232, "sampling/importance_sampling_ratio/min": 0.23671342432498932, "sampling/importance_sampling_ratio/mean": 1.0233737230300903, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.241538181900978, "clip_ratio/low_mean": 0.01886792480945587, "clip_ratio/low_min": 0.01886792480945587, "clip_ratio/high_mean": 0.15008282009512186, "clip_ratio/high_max": 0.15008282009512186, "clip_ratio/region_mean": 0.16895074490457773, "reward_total_mean": 0.7346892952919006, "reward_meter_mean": 0.9180644750595093, "reward_meter_std": 0.1755811870098114, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9243527054786682, "reward_repeat_soft_std": 0.06503698229789734, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7346892952919006, "reward_total_composite_std": 0.0747465193271637} {"timestamp_utc": "2026-04-13T00:30:29Z", "mode": "train", "global_step": 908, "epoch": 0.09121044701155198, "loss": -0.0058, "grad_norm": 11.275321960449219, "learning_rate": 7.251515151515151e-06, "num_tokens": 1649978.0, "completions/mean_length": 60.875, "completions/min_length": 35.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9944349527359009, "rewards/meter/std": 0.0025885493960231543, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7857185006141663, "rewards/repeat_soft/std": 0.1568850725889206, "rewards/judge_quality/mean": 0.44875001907348633, "rewards/judge_quality/std": 0.21256513893604279, "rewards/total_composite/mean": 0.763817548751831, "rewards/total_composite/std": 0.08421817421913147, "reward": 0.763817548751831, "reward_std": 0.08421817421913147, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14391811192035675, "sampling/sampling_logp_difference/max": 1.9077175855636597, "sampling/importance_sampling_ratio/min": 0.14841875433921814, "sampling/importance_sampling_ratio/mean": 1.0283838510513306, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1480392664670944, "clip_ratio/low_mean": 0.016418875311501324, "clip_ratio/low_min": 0.016418875311501324, "clip_ratio/high_mean": 0.09059870336204767, "clip_ratio/high_max": 0.09059870336204767, "clip_ratio/region_mean": 0.107017578673549, "reward_total_mean": 0.763817548751831, "reward_meter_mean": 0.9944349527359009, "reward_meter_std": 0.0025885493960231543, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7857185006141663, "reward_repeat_soft_std": 0.1568850725889206, "reward_judge_quality_mean": 0.44875001907348633, "reward_judge_quality_std": 0.21256513893604279, "reward_total_composite_mean": 0.763817548751831, "reward_total_composite_std": 0.08421817421913147} {"timestamp_utc": "2026-04-13T00:30:35Z", "mode": "train", "global_step": 909, "epoch": 0.09131089904570568, "loss": 0.1286, "grad_norm": 14.184093475341797, "learning_rate": 7.2484848484848495e-06, "num_tokens": 1651496.0, "completions/mean_length": 30.75, "completions/min_length": 24.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.75, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.9019205570220947, "rewards/meter/std": 0.07605462521314621, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9270409345626831, "rewards/repeat_soft/std": 0.07701418548822403, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.8143182992935181, "rewards/total_composite/std": 0.08913944661617279, "reward": 0.8143182992935181, "reward_std": 0.08913944661617279, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12805190682411194, "sampling/sampling_logp_difference/max": 1.0017821788787842, "sampling/importance_sampling_ratio/min": 0.36722439527511597, "sampling/importance_sampling_ratio/mean": 1.009281039237976, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7550113350152969, "clip_ratio/low_mean": 0.10043546184897423, "clip_ratio/low_min": 0.10043546184897423, "clip_ratio/high_mean": 0.033518518321216106, "clip_ratio/high_max": 0.033518518321216106, "clip_ratio/region_mean": 0.13395398017019033, "reward_total_mean": 0.8143182992935181, "reward_meter_mean": 0.9019205570220947, "reward_meter_std": 0.07605462521314621, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9270409345626831, "reward_repeat_soft_std": 0.07701418548822403, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.8143182992935181, "reward_total_composite_std": 0.08913944661617279} {"timestamp_utc": "2026-04-13T00:30:42Z", "mode": "train", "global_step": 910, "epoch": 0.09141135107985937, "loss": -0.0132, "grad_norm": 14.943204879760742, "learning_rate": 7.245454545454546e-06, "num_tokens": 1653215.0, "completions/mean_length": 42.875, "completions/min_length": 38.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.875, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.7074174880981445, "rewards/meter/std": 0.3108137547969818, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9695740938186646, "rewards/repeat_soft/std": 0.03494240716099739, "rewards/judge_quality/mean": 0.5562500357627869, "rewards/judge_quality/std": 0.31717222929000854, "rewards/total_composite/mean": 0.6398844718933105, "rewards/total_composite/std": 0.29692187905311584, "reward": 0.6398844718933105, "reward_std": 0.29692187905311584, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17568734288215637, "sampling/sampling_logp_difference/max": 1.6687536239624023, "sampling/importance_sampling_ratio/min": 0.18848183751106262, "sampling/importance_sampling_ratio/mean": 1.0466532707214355, "sampling/importance_sampling_ratio/max": 1.781883716583252, "entropy": 1.9399268627166748, "clip_ratio/low_mean": 0.031685903668403625, "clip_ratio/low_min": 0.031685903668403625, "clip_ratio/high_mean": 0.10312654636800289, "clip_ratio/high_max": 0.10312654636800289, "clip_ratio/region_mean": 0.13481245003640652, "reward_total_mean": 0.6398844718933105, "reward_meter_mean": 0.7074174880981445, "reward_meter_std": 0.3108137547969818, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9695740938186646, "reward_repeat_soft_std": 0.03494240716099739, "reward_judge_quality_mean": 0.5562500357627869, "reward_judge_quality_std": 0.31717222929000854, "reward_total_composite_mean": 0.6398844718933105, "reward_total_composite_std": 0.29692187905311584} {"timestamp_utc": "2026-04-13T00:30:49Z", "mode": "train", "global_step": 911, "epoch": 0.09151180311401307, "loss": -0.0315, "grad_norm": 16.070417404174805, "learning_rate": 7.242424242424243e-06, "num_tokens": 1654862.0, "completions/mean_length": 26.875, "completions/min_length": 23.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.875, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.7741059064865112, "rewards/meter/std": 0.33563074469566345, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9471848011016846, "rewards/repeat_soft/std": 0.05794452130794525, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.8315661549568176, "rewards/total_composite/std": 0.15923351049423218, "reward": 0.8315661549568176, "reward_std": 0.1592334806919098, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14363960921764374, "sampling/sampling_logp_difference/max": 1.3733243942260742, "sampling/importance_sampling_ratio/min": 0.2532636225223541, "sampling/importance_sampling_ratio/mean": 1.021960735321045, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9870025292038918, "clip_ratio/low_mean": 0.048812830820679665, "clip_ratio/low_min": 0.048812830820679665, "clip_ratio/high_mean": 0.0557413618080318, "clip_ratio/high_max": 0.0557413618080318, "clip_ratio/region_mean": 0.10455419262871146, "reward_total_mean": 0.8315661549568176, "reward_meter_mean": 0.7741059064865112, "reward_meter_std": 0.33563074469566345, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9471848011016846, "reward_repeat_soft_std": 0.05794452130794525, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.8315661549568176, "reward_total_composite_std": 0.15923351049423218} {"timestamp_utc": "2026-04-13T00:30:57Z", "mode": "train", "global_step": 912, "epoch": 0.09161225514816675, "loss": 0.0934, "grad_norm": 23.926605224609375, "learning_rate": 7.2393939393939404e-06, "num_tokens": 1656451.0, "completions/mean_length": 24.625, "completions/min_length": 22.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.625, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.5067636966705322, "rewards/meter/std": 0.3432392179965973, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9503787755966187, "rewards/repeat_soft/std": 0.03874915465712547, "rewards/judge_quality/mean": 0.38874998688697815, "rewards/judge_quality/std": 0.08675704896450043, "rewards/total_composite/mean": 0.5897065997123718, "rewards/total_composite/std": 0.14830996096134186, "reward": 0.5897065997123718, "reward_std": 0.14830996096134186, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1872318983078003, "sampling/sampling_logp_difference/max": 1.5367603302001953, "sampling/importance_sampling_ratio/min": 0.215076744556427, "sampling/importance_sampling_ratio/mean": 1.0351061820983887, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0123384222388268, "clip_ratio/low_mean": 0.11095467302948236, "clip_ratio/low_min": 0.11095467302948236, "clip_ratio/high_mean": 0.05409090965986252, "clip_ratio/high_max": 0.05409090965986252, "clip_ratio/region_mean": 0.16504558268934488, "reward_total_mean": 0.5897065997123718, "reward_meter_mean": 0.5067636966705322, "reward_meter_std": 0.3432392179965973, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9503787755966187, "reward_repeat_soft_std": 0.03874915465712547, "reward_judge_quality_mean": 0.38874998688697815, "reward_judge_quality_std": 0.08675704896450043, "reward_total_composite_mean": 0.5897065997123718, "reward_total_composite_std": 0.14830996096134186} {"timestamp_utc": "2026-04-13T00:31:05Z", "mode": "train", "global_step": 913, "epoch": 0.09171270718232044, "loss": 0.022, "grad_norm": 16.731578826904297, "learning_rate": 7.236363636363637e-06, "num_tokens": 1657928.0, "completions/mean_length": 36.625, "completions/min_length": 34.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.625, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.9237072467803955, "rewards/meter/std": 0.18568170070648193, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8839038610458374, "rewards/repeat_soft/std": 0.08524762094020844, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.8010586500167847, "rewards/total_composite/std": 0.10690438002347946, "reward": 0.8010586500167847, "reward_std": 0.10690437257289886, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14673234522342682, "sampling/sampling_logp_difference/max": 1.5223631858825684, "sampling/importance_sampling_ratio/min": 0.21819564700126648, "sampling/importance_sampling_ratio/mean": 1.0164819955825806, "sampling/importance_sampling_ratio/max": 1.9156173467636108, "entropy": 1.114958181977272, "clip_ratio/low_mean": 0.009868420660495758, "clip_ratio/low_min": 0.009868420660495758, "clip_ratio/high_mean": 0.09344289940781891, "clip_ratio/high_max": 0.09344289940781891, "clip_ratio/region_mean": 0.10331132006831467, "reward_total_mean": 0.8010586500167847, "reward_meter_mean": 0.9237072467803955, "reward_meter_std": 0.18568170070648193, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8839038610458374, "reward_repeat_soft_std": 0.08524762094020844, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.8010586500167847, "reward_total_composite_std": 0.10690438002347946} {"timestamp_utc": "2026-04-13T00:31:19Z", "mode": "train", "global_step": 914, "epoch": 0.09181315921647414, "loss": -0.1194, "grad_norm": 3.289896249771118, "learning_rate": 7.233333333333334e-06, "num_tokens": 1659559.0, "completions/mean_length": 113.875, "completions/min_length": 52.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 57.000003814697266, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.8222301602363586, "rewards/meter/std": 0.34721848368644714, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9728337526321411, "rewards/repeat_soft/std": 0.046736638993024826, "rewards/judge_quality/mean": 0.4087499976158142, "rewards/judge_quality/std": 0.2527809739112854, "rewards/total_composite/mean": 0.6510475277900696, "rewards/total_composite/std": 0.31486865878105164, "reward": 0.6510475277900696, "reward_std": 0.31486862897872925, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12665830552577972, "sampling/sampling_logp_difference/max": 1.3334097862243652, "sampling/importance_sampling_ratio/min": 0.2635769844055176, "sampling/importance_sampling_ratio/mean": 1.0138317346572876, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7906329780817032, "clip_ratio/low_mean": 0.037265434861183167, "clip_ratio/low_min": 0.037265434861183167, "clip_ratio/high_mean": 0.09185498114675283, "clip_ratio/high_max": 0.09185498114675283, "clip_ratio/region_mean": 0.129120416007936, "reward_total_mean": 0.6510475277900696, "reward_meter_mean": 0.8222301602363586, "reward_meter_std": 0.34721848368644714, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9728337526321411, "reward_repeat_soft_std": 0.046736638993024826, "reward_judge_quality_mean": 0.4087499976158142, "reward_judge_quality_std": 0.2527809739112854, "reward_total_composite_mean": 0.6510475277900696, "reward_total_composite_std": 0.31486865878105164} {"timestamp_utc": "2026-04-13T00:31:26Z", "mode": "train", "global_step": 915, "epoch": 0.09191361125062783, "loss": -0.016, "grad_norm": 11.995733261108398, "learning_rate": 7.2303030303030305e-06, "num_tokens": 1661264.0, "completions/mean_length": 50.125, "completions/min_length": 45.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.125, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.888860821723938, "rewards/meter/std": 0.11366664618253708, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9767439961433411, "rewards/repeat_soft/std": 0.011139142327010632, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.7783492803573608, "rewards/total_composite/std": 0.09180819243192673, "reward": 0.7783492803573608, "reward_std": 0.09180818498134613, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14059285819530487, "sampling/sampling_logp_difference/max": 1.8617632389068604, "sampling/importance_sampling_ratio/min": 0.1553983837366104, "sampling/importance_sampling_ratio/mean": 0.9897767305374146, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8310428857803345, "clip_ratio/low_mean": 0.10415356489829719, "clip_ratio/low_min": 0.10415356489829719, "clip_ratio/high_mean": 0.03774590138345957, "clip_ratio/high_max": 0.03774590138345957, "clip_ratio/region_mean": 0.14189946628175676, "reward_total_mean": 0.7783492803573608, "reward_meter_mean": 0.888860821723938, "reward_meter_std": 0.11366664618253708, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9767439961433411, "reward_repeat_soft_std": 0.011139142327010632, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.7783492803573608, "reward_total_composite_std": 0.09180819243192673} {"timestamp_utc": "2026-04-13T00:31:34Z", "mode": "train", "global_step": 916, "epoch": 0.09201406328478151, "loss": 0.0596, "grad_norm": 61.06100845336914, "learning_rate": 7.227272727272729e-06, "num_tokens": 1662755.0, "completions/mean_length": 25.375, "completions/min_length": 23.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.375, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.8392468094825745, "rewards/meter/std": 0.2585483491420746, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9419333934783936, "rewards/repeat_soft/std": 0.05288024991750717, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7501044273376465, "rewards/total_composite/std": 0.11998241394758224, "reward": 0.7501044273376465, "reward_std": 0.11998242139816284, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16409102082252502, "sampling/sampling_logp_difference/max": 1.6285743713378906, "sampling/importance_sampling_ratio/min": 0.1962091028690338, "sampling/importance_sampling_ratio/mean": 0.978768527507782, "sampling/importance_sampling_ratio/max": 1.9323959350585938, "entropy": 0.7957031056284904, "clip_ratio/low_mean": 0.014999999664723873, "clip_ratio/low_min": 0.014999999664723873, "clip_ratio/high_mean": 0.14223304111510515, "clip_ratio/high_max": 0.14223304111510515, "clip_ratio/region_mean": 0.15723304077982903, "reward_total_mean": 0.7501044273376465, "reward_meter_mean": 0.8392468094825745, "reward_meter_std": 0.2585483491420746, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9419333934783936, "reward_repeat_soft_std": 0.05288024991750717, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7501044273376465, "reward_total_composite_std": 0.11998241394758224} {"timestamp_utc": "2026-04-13T00:31:41Z", "mode": "train", "global_step": 917, "epoch": 0.0921145153189352, "loss": 0.0538, "grad_norm": 18.975627899169922, "learning_rate": 7.224242424242425e-06, "num_tokens": 1664137.0, "completions/mean_length": 32.75, "completions/min_length": 29.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.75, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.8330504298210144, "rewards/meter/std": 0.2899457812309265, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9856334924697876, "rewards/repeat_soft/std": 0.02391321398317814, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.11310552060604095, "rewards/total_composite/mean": 0.765186071395874, "rewards/total_composite/std": 0.13788874447345734, "reward": 0.765186071395874, "reward_std": 0.13788874447345734, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17222903668880463, "sampling/sampling_logp_difference/max": 1.3195312023162842, "sampling/importance_sampling_ratio/min": 0.296902596950531, "sampling/importance_sampling_ratio/mean": 1.051896333694458, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4448612183332443, "clip_ratio/low_mean": 0.02250445680692792, "clip_ratio/low_min": 0.02250445680692792, "clip_ratio/high_mean": 0.13738290406763554, "clip_ratio/high_max": 0.13738290406763554, "clip_ratio/region_mean": 0.15988736087456346, "reward_total_mean": 0.765186071395874, "reward_meter_mean": 0.8330504298210144, "reward_meter_std": 0.2899457812309265, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9856334924697876, "reward_repeat_soft_std": 0.02391321398317814, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.11310552060604095, "reward_total_composite_mean": 0.765186071395874, "reward_total_composite_std": 0.13788874447345734} {"timestamp_utc": "2026-04-13T00:31:48Z", "mode": "train", "global_step": 918, "epoch": 0.0922149673530889, "loss": 0.0032, "grad_norm": 14.925040245056152, "learning_rate": 7.221212121212122e-06, "num_tokens": 1665665.0, "completions/mean_length": 39.0, "completions/min_length": 31.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.8069719076156616, "rewards/meter/std": 0.3223312199115753, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9828983545303345, "rewards/repeat_soft/std": 0.02150745876133442, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.6998021602630615, "rewards/total_composite/std": 0.15847885608673096, "reward": 0.6998021602630615, "reward_std": 0.15847887098789215, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18962980806827545, "sampling/sampling_logp_difference/max": 1.2771339416503906, "sampling/importance_sampling_ratio/min": 0.2788352966308594, "sampling/importance_sampling_ratio/mean": 1.0270253419876099, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5787113681435585, "clip_ratio/low_mean": 0.042220113798975945, "clip_ratio/low_min": 0.042220113798975945, "clip_ratio/high_mean": 0.12450866727158427, "clip_ratio/high_max": 0.12450866727158427, "clip_ratio/region_mean": 0.16672878107056022, "reward_total_mean": 0.6998021602630615, "reward_meter_mean": 0.8069719076156616, "reward_meter_std": 0.3223312199115753, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9828983545303345, "reward_repeat_soft_std": 0.02150745876133442, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.6998021602630615, "reward_total_composite_std": 0.15847885608673096} {"timestamp_utc": "2026-04-13T00:31:56Z", "mode": "train", "global_step": 919, "epoch": 0.0923154193872426, "loss": 0.0709, "grad_norm": 12.571917533874512, "learning_rate": 7.218181818181819e-06, "num_tokens": 1667236.0, "completions/mean_length": 55.375, "completions/min_length": 48.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.375, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9449951648712158, "rewards/meter/std": 0.10540029406547546, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.968257486820221, "rewards/repeat_soft/std": 0.03290222957730293, "rewards/judge_quality/mean": 0.4012500047683716, "rewards/judge_quality/std": 0.22793719172477722, "rewards/total_composite/mean": 0.7924485206604004, "rewards/total_composite/std": 0.09642162173986435, "reward": 0.7924485206604004, "reward_std": 0.09642162173986435, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14849385619163513, "sampling/sampling_logp_difference/max": 1.246004581451416, "sampling/importance_sampling_ratio/min": 0.28765180706977844, "sampling/importance_sampling_ratio/mean": 1.0237138271331787, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9806299284100533, "clip_ratio/low_mean": 0.06505720503628254, "clip_ratio/low_min": 0.06505720503628254, "clip_ratio/high_mean": 0.06734173558652401, "clip_ratio/high_max": 0.06734173558652401, "clip_ratio/region_mean": 0.13239894062280655, "reward_total_mean": 0.7924485206604004, "reward_meter_mean": 0.9449951648712158, "reward_meter_std": 0.10540029406547546, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.968257486820221, "reward_repeat_soft_std": 0.03290222957730293, "reward_judge_quality_mean": 0.4012500047683716, "reward_judge_quality_std": 0.22793719172477722, "reward_total_composite_mean": 0.7924485206604004, "reward_total_composite_std": 0.09642162173986435} {"timestamp_utc": "2026-04-13T00:32:04Z", "mode": "train", "global_step": 920, "epoch": 0.09241587142139629, "loss": 0.0518, "grad_norm": 14.103556632995605, "learning_rate": 7.215151515151516e-06, "num_tokens": 1669037.0, "completions/mean_length": 57.125, "completions/min_length": 53.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.125, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.5328102111816406, "rewards/meter/std": 0.3785361647605896, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9014208912849426, "rewards/repeat_soft/std": 0.08334723860025406, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5684067010879517, "rewards/total_composite/std": 0.17504309117794037, "reward": 0.5684067010879517, "reward_std": 0.17504309117794037, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1757451593875885, "sampling/sampling_logp_difference/max": 1.6477017402648926, "sampling/importance_sampling_ratio/min": 0.1924917846918106, "sampling/importance_sampling_ratio/mean": 1.0144767761230469, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.378240704536438, "clip_ratio/low_mean": 0.054167880676686764, "clip_ratio/low_min": 0.054167880676686764, "clip_ratio/high_mean": 0.07327792886644602, "clip_ratio/high_max": 0.07327792886644602, "clip_ratio/region_mean": 0.12744580954313278, "reward_total_mean": 0.5684067010879517, "reward_meter_mean": 0.5328102111816406, "reward_meter_std": 0.3785361647605896, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9014208912849426, "reward_repeat_soft_std": 0.08334723860025406, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5684067010879517, "reward_total_composite_std": 0.17504309117794037} {"timestamp_utc": "2026-04-13T00:32:11Z", "mode": "train", "global_step": 921, "epoch": 0.09251632345554997, "loss": -0.0479, "grad_norm": 15.405622482299805, "learning_rate": 7.212121212121212e-06, "num_tokens": 1670532.0, "completions/mean_length": 33.875, "completions/min_length": 25.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.875, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.5047249794006348, "rewards/meter/std": 0.4019019603729248, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9768311381340027, "rewards/repeat_soft/std": 0.01726122386753559, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.6394343376159668, "rewards/total_composite/std": 0.18507017195224762, "reward": 0.6394343376159668, "reward_std": 0.18507017195224762, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14135147631168365, "sampling/sampling_logp_difference/max": 1.6292152404785156, "sampling/importance_sampling_ratio/min": 0.19608339667320251, "sampling/importance_sampling_ratio/mean": 1.030396819114685, "sampling/importance_sampling_ratio/max": 1.8882681131362915, "entropy": 1.1103454753756523, "clip_ratio/low_mean": 0.07560246624052525, "clip_ratio/low_min": 0.07560246624052525, "clip_ratio/high_mean": 0.08988095633685589, "clip_ratio/high_max": 0.08988095633685589, "clip_ratio/region_mean": 0.16548342257738113, "reward_total_mean": 0.6394343376159668, "reward_meter_mean": 0.5047249794006348, "reward_meter_std": 0.4019019603729248, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9768311381340027, "reward_repeat_soft_std": 0.01726122386753559, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.6394343376159668, "reward_total_composite_std": 0.18507017195224762} {"timestamp_utc": "2026-04-13T00:32:19Z", "mode": "train", "global_step": 922, "epoch": 0.09261677548970366, "loss": 0.0327, "grad_norm": 21.048376083374023, "learning_rate": 7.2090909090909104e-06, "num_tokens": 1672338.0, "completions/mean_length": 59.75, "completions/min_length": 52.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.75, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.49346333742141724, "rewards/meter/std": 0.39630794525146484, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9918150305747986, "rewards/repeat_soft/std": 0.011093336157500744, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.6769275069236755, "rewards/total_composite/std": 0.19787752628326416, "reward": 0.6769275069236755, "reward_std": 0.19787752628326416, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17715014517307281, "sampling/sampling_logp_difference/max": 5.6129536628723145, "sampling/importance_sampling_ratio/min": 0.0036502715665847063, "sampling/importance_sampling_ratio/mean": 0.9871840476989746, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6757016554474831, "clip_ratio/low_mean": 0.05014467611908913, "clip_ratio/low_min": 0.05014467611908913, "clip_ratio/high_mean": 0.07093494571745396, "clip_ratio/high_max": 0.07093494571745396, "clip_ratio/region_mean": 0.12107962183654308, "reward_total_mean": 0.6769275069236755, "reward_meter_mean": 0.49346333742141724, "reward_meter_std": 0.39630794525146484, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9918150305747986, "reward_repeat_soft_std": 0.011093336157500744, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.6769275069236755, "reward_total_composite_std": 0.19787752628326416} {"timestamp_utc": "2026-04-13T00:32:27Z", "mode": "train", "global_step": 923, "epoch": 0.09271722752385736, "loss": 0.0723, "grad_norm": 15.884596824645996, "learning_rate": 7.206060606060606e-06, "num_tokens": 1673700.0, "completions/mean_length": 33.25, "completions/min_length": 27.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.25, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.7171269655227661, "rewards/meter/std": 0.3654637932777405, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9903541803359985, "rewards/repeat_soft/std": 0.011319226585328579, "rewards/judge_quality/mean": 0.5024999976158142, "rewards/judge_quality/std": 0.1348809152841568, "rewards/total_composite/mean": 0.7224925756454468, "rewards/total_composite/std": 0.182597354054451, "reward": 0.7224925756454468, "reward_std": 0.1825973391532898, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14046059548854828, "sampling/sampling_logp_difference/max": 1.2903175354003906, "sampling/importance_sampling_ratio/min": 0.27518340945243835, "sampling/importance_sampling_ratio/mean": 1.0152889490127563, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0106092169880867, "clip_ratio/low_mean": 0.05743534443899989, "clip_ratio/low_min": 0.05743534443899989, "clip_ratio/high_mean": 0.07148559764027596, "clip_ratio/high_max": 0.07148559764027596, "clip_ratio/region_mean": 0.12892094207927585, "reward_total_mean": 0.7224925756454468, "reward_meter_mean": 0.7171269655227661, "reward_meter_std": 0.3654637932777405, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9903541803359985, "reward_repeat_soft_std": 0.011319226585328579, "reward_judge_quality_mean": 0.5024999976158142, "reward_judge_quality_std": 0.1348809152841568, "reward_total_composite_mean": 0.7224925756454468, "reward_total_composite_std": 0.182597354054451} {"timestamp_utc": "2026-04-13T00:32:34Z", "mode": "train", "global_step": 924, "epoch": 0.09281767955801105, "loss": -0.0116, "grad_norm": 14.427873611450195, "learning_rate": 7.203030303030304e-06, "num_tokens": 1675168.0, "completions/mean_length": 32.5, "completions/min_length": 28.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.5, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.3721855878829956, "rewards/meter/std": 0.3883007764816284, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9719866514205933, "rewards/repeat_soft/std": 0.038132473826408386, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.5530571937561035, "rewards/total_composite/std": 0.19186583161354065, "reward": 0.5530571937561035, "reward_std": 0.19186580181121826, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18517735600471497, "sampling/sampling_logp_difference/max": 1.520493984222412, "sampling/importance_sampling_ratio/min": 0.21860387921333313, "sampling/importance_sampling_ratio/mean": 1.0339854955673218, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.396124079823494, "clip_ratio/low_mean": 0.10464728064835072, "clip_ratio/low_min": 0.10464728064835072, "clip_ratio/high_mean": 0.07319180108606815, "clip_ratio/high_max": 0.07319180108606815, "clip_ratio/region_mean": 0.17783908173441887, "reward_total_mean": 0.5530571937561035, "reward_meter_mean": 0.3721855878829956, "reward_meter_std": 0.3883007764816284, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9719866514205933, "reward_repeat_soft_std": 0.038132473826408386, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.5530571937561035, "reward_total_composite_std": 0.19186583161354065} {"timestamp_utc": "2026-04-13T00:32:40Z", "mode": "train", "global_step": 925, "epoch": 0.09291813159216473, "loss": 0.0008, "grad_norm": 13.924239158630371, "learning_rate": 7.2000000000000005e-06, "num_tokens": 1676881.0, "completions/mean_length": 36.125, "completions/min_length": 32.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.125, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.8471813201904297, "rewards/meter/std": 0.3295278549194336, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9826076030731201, "rewards/repeat_soft/std": 0.02140755020081997, "rewards/judge_quality/mean": 0.6187499761581421, "rewards/judge_quality/std": 0.24976776540279388, "rewards/total_composite/mean": 0.815117359161377, "rewards/total_composite/std": 0.18707342445850372, "reward": 0.815117359161377, "reward_std": 0.18707340955734253, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16532623767852783, "sampling/sampling_logp_difference/max": 1.821742057800293, "sampling/importance_sampling_ratio/min": 0.16174374520778656, "sampling/importance_sampling_ratio/mean": 1.0311278104782104, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1825122609734535, "clip_ratio/low_mean": 0.04904746077954769, "clip_ratio/low_min": 0.04904746077954769, "clip_ratio/high_mean": 0.101846594363451, "clip_ratio/high_max": 0.101846594363451, "clip_ratio/region_mean": 0.1508940551429987, "reward_total_mean": 0.815117359161377, "reward_meter_mean": 0.8471813201904297, "reward_meter_std": 0.3295278549194336, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9826076030731201, "reward_repeat_soft_std": 0.02140755020081997, "reward_judge_quality_mean": 0.6187499761581421, "reward_judge_quality_std": 0.24976776540279388, "reward_total_composite_mean": 0.815117359161377, "reward_total_composite_std": 0.18707342445850372} {"timestamp_utc": "2026-04-13T00:32:49Z", "mode": "train", "global_step": 926, "epoch": 0.09301858362631843, "loss": -0.1287, "grad_norm": 7.678193092346191, "learning_rate": 7.196969696969698e-06, "num_tokens": 1679519.0, "completions/mean_length": 112.75, "completions/min_length": 70.0, "completions/max_length": 154.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.75, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 154.0, "rewards/meter/mean": 0.8903652429580688, "rewards/meter/std": 0.2736362814903259, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8443437218666077, "rewards/repeat_soft/std": 0.14593550562858582, "rewards/judge_quality/mean": 0.3037499785423279, "rewards/judge_quality/std": 0.13362392783164978, "rewards/total_composite/mean": 0.7012237310409546, "rewards/total_composite/std": 0.1213720440864563, "reward": 0.7012237310409546, "reward_std": 0.1213720440864563, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13184583187103271, "sampling/sampling_logp_difference/max": 1.9830207824707031, "sampling/importance_sampling_ratio/min": 0.13765278458595276, "sampling/importance_sampling_ratio/mean": 1.0230960845947266, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0352565199136734, "clip_ratio/low_mean": 0.036265598610043526, "clip_ratio/low_min": 0.036265598610043526, "clip_ratio/high_mean": 0.08684017322957516, "clip_ratio/high_max": 0.08684017322957516, "clip_ratio/region_mean": 0.12310577183961868, "reward_total_mean": 0.7012237310409546, "reward_meter_mean": 0.8903652429580688, "reward_meter_std": 0.2736362814903259, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8443437218666077, "reward_repeat_soft_std": 0.14593550562858582, "reward_judge_quality_mean": 0.3037499785423279, "reward_judge_quality_std": 0.13362392783164978, "reward_total_composite_mean": 0.7012237310409546, "reward_total_composite_std": 0.1213720440864563} {"timestamp_utc": "2026-04-13T00:33:01Z", "mode": "train", "global_step": 927, "epoch": 0.09311903566047212, "loss": -0.1037, "grad_norm": 3.1575708389282227, "learning_rate": 7.193939393939394e-06, "num_tokens": 1681341.0, "completions/mean_length": 163.75, "completions/min_length": 42.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 47.66666793823242, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.6725602149963379, "rewards/meter/std": 0.4328930079936981, "rewards/count_adherence/mean": 0.65625, "rewards/count_adherence/std": 0.18600596487522125, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.948337197303772, "rewards/repeat_soft/std": 0.06198639050126076, "rewards/judge_quality/mean": 0.32749998569488525, "rewards/judge_quality/std": 0.17127670347690582, "rewards/total_composite/mean": 0.491102010011673, "rewards/total_composite/std": 0.34003978967666626, "reward": 0.491102010011673, "reward_std": 0.34003978967666626, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21798738837242126, "sampling/sampling_logp_difference/max": 1.4044380187988281, "sampling/importance_sampling_ratio/min": 0.24550499022006989, "sampling/importance_sampling_ratio/mean": 1.0432353019714355, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3265634775161743, "clip_ratio/low_mean": 0.05772946774959564, "clip_ratio/low_min": 0.05772946774959564, "clip_ratio/high_mean": 0.12290903832763433, "clip_ratio/high_max": 0.12290903832763433, "clip_ratio/region_mean": 0.18063850607722998, "reward_total_mean": 0.491102010011673, "reward_meter_mean": 0.6725602149963379, "reward_meter_std": 0.4328930079936981, "reward_count_adherence_mean": 0.65625, "reward_count_adherence_std": 0.18600596487522125, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.948337197303772, "reward_repeat_soft_std": 0.06198639050126076, "reward_judge_quality_mean": 0.32749998569488525, "reward_judge_quality_std": 0.17127670347690582, "reward_total_composite_mean": 0.491102010011673, "reward_total_composite_std": 0.34003978967666626} {"timestamp_utc": "2026-04-13T00:33:13Z", "mode": "train", "global_step": 928, "epoch": 0.09321948769462582, "loss": -0.1373, "grad_norm": 3.612257719039917, "learning_rate": 7.1909090909090914e-06, "num_tokens": 1683047.0, "completions/mean_length": 113.25, "completions/min_length": 44.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 56.28571701049805, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.6357231736183167, "rewards/meter/std": 0.4672423303127289, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9434494972229004, "rewards/repeat_soft/std": 0.08632058650255203, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.13845446705818176, "rewards/total_composite/mean": 0.6041703820228577, "rewards/total_composite/std": 0.29647591710090637, "reward": 0.6041703820228577, "reward_std": 0.29647591710090637, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13765650987625122, "sampling/sampling_logp_difference/max": 1.6739563941955566, "sampling/importance_sampling_ratio/min": 0.18750375509262085, "sampling/importance_sampling_ratio/mean": 1.0149744749069214, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.089166522026062, "clip_ratio/low_mean": 0.0358851682394743, "clip_ratio/low_min": 0.0358851682394743, "clip_ratio/high_mean": 0.08001700323075056, "clip_ratio/high_max": 0.08001700323075056, "clip_ratio/region_mean": 0.11590217147022486, "reward_total_mean": 0.6041703820228577, "reward_meter_mean": 0.6357231736183167, "reward_meter_std": 0.4672423303127289, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9434494972229004, "reward_repeat_soft_std": 0.08632058650255203, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.13845446705818176, "reward_total_composite_mean": 0.6041703820228577, "reward_total_composite_std": 0.29647591710090637} {"timestamp_utc": "2026-04-13T00:33:20Z", "mode": "train", "global_step": 929, "epoch": 0.09331993972877951, "loss": 0.0053, "grad_norm": 30.0685977935791, "learning_rate": 7.187878787878788e-06, "num_tokens": 1684464.0, "completions/mean_length": 19.125, "completions/min_length": 17.0, "completions/max_length": 21.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.125, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 21.0, "rewards/meter/mean": 0.8745841383934021, "rewards/meter/std": 0.3275156021118164, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8968881368637085, "rewards/repeat_soft/std": 0.12212178111076355, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.19255799055099487, "rewards/total_composite/mean": 0.7750016450881958, "rewards/total_composite/std": 0.16364061832427979, "reward": 0.7750016450881958, "reward_std": 0.16364061832427979, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14082618057727814, "sampling/sampling_logp_difference/max": 1.4128093719482422, "sampling/importance_sampling_ratio/min": 0.2434583604335785, "sampling/importance_sampling_ratio/mean": 1.034049391746521, "sampling/importance_sampling_ratio/max": 1.6791303157806396, "entropy": 1.3823945224285126, "clip_ratio/low_mean": 0.027046783827245235, "clip_ratio/low_min": 0.027046783827245235, "clip_ratio/high_mean": 0.13879091059789062, "clip_ratio/high_max": 0.13879091059789062, "clip_ratio/region_mean": 0.16583769442513585, "reward_total_mean": 0.7750016450881958, "reward_meter_mean": 0.8745841383934021, "reward_meter_std": 0.3275156021118164, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8968881368637085, "reward_repeat_soft_std": 0.12212178111076355, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.19255799055099487, "reward_total_composite_mean": 0.7750016450881958, "reward_total_composite_std": 0.16364061832427979} {"timestamp_utc": "2026-04-13T00:33:31Z", "mode": "train", "global_step": 930, "epoch": 0.0934203917629332, "loss": 0.0069, "grad_norm": 21.998762130737305, "learning_rate": 7.184848484848486e-06, "num_tokens": 1686058.0, "completions/mean_length": 34.25, "completions/min_length": 31.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.25, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.7093390822410583, "rewards/meter/std": 0.4189549386501312, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9792742133140564, "rewards/repeat_soft/std": 0.017144931480288506, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.7516299486160278, "rewards/total_composite/std": 0.19298140704631805, "reward": 0.7516299486160278, "reward_std": 0.19298140704631805, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18243014812469482, "sampling/sampling_logp_difference/max": 1.5932674407958984, "sampling/importance_sampling_ratio/min": 0.2032603770494461, "sampling/importance_sampling_ratio/mean": 1.0327638387680054, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.465874344110489, "clip_ratio/low_mean": 0.05748932622373104, "clip_ratio/low_min": 0.05748932622373104, "clip_ratio/high_mean": 0.09214913751929998, "clip_ratio/high_max": 0.09214913751929998, "clip_ratio/region_mean": 0.14963846374303102, "reward_total_mean": 0.7516299486160278, "reward_meter_mean": 0.7093390822410583, "reward_meter_std": 0.4189549386501312, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9792742133140564, "reward_repeat_soft_std": 0.017144931480288506, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.7516299486160278, "reward_total_composite_std": 0.19298140704631805} {"timestamp_utc": "2026-04-13T00:33:38Z", "mode": "train", "global_step": 931, "epoch": 0.09352084379708689, "loss": -0.0343, "grad_norm": 31.199073791503906, "learning_rate": 7.181818181818182e-06, "num_tokens": 1687390.0, "completions/mean_length": 17.5, "completions/min_length": 14.0, "completions/max_length": 22.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 17.5, "completions/min_terminated_length": 14.0, "completions/max_terminated_length": 22.0, "rewards/meter/mean": 0.730293869972229, "rewards/meter/std": 0.43777939677238464, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9444704651832581, "rewards/repeat_soft/std": 0.05099513754248619, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.25587037205696106, "rewards/total_composite/mean": 0.7564542889595032, "rewards/total_composite/std": 0.24075813591480255, "reward": 0.7564542889595032, "reward_std": 0.24075813591480255, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19354146718978882, "sampling/sampling_logp_difference/max": 1.7371554374694824, "sampling/importance_sampling_ratio/min": 0.17602039873600006, "sampling/importance_sampling_ratio/mean": 1.024958848953247, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9118326976895332, "clip_ratio/low_mean": 0.04241071455180645, "clip_ratio/low_min": 0.04241071455180645, "clip_ratio/high_mean": 0.13076097890734673, "clip_ratio/high_max": 0.13076097890734673, "clip_ratio/region_mean": 0.17317169345915318, "reward_total_mean": 0.7564542889595032, "reward_meter_mean": 0.730293869972229, "reward_meter_std": 0.43777939677238464, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9444704651832581, "reward_repeat_soft_std": 0.05099513754248619, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.25587037205696106, "reward_total_composite_mean": 0.7564542889595032, "reward_total_composite_std": 0.24075813591480255} {"timestamp_utc": "2026-04-13T00:33:51Z", "mode": "train", "global_step": 932, "epoch": 0.09362129583124058, "loss": -0.1346, "grad_norm": 3.280212163925171, "learning_rate": 7.17878787878788e-06, "num_tokens": 1689360.0, "completions/mean_length": 117.25, "completions/min_length": 44.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 60.857147216796875, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.7598826289176941, "rewards/meter/std": 0.40752553939819336, "rewards/count_adherence/mean": 0.675000011920929, "rewards/count_adherence/std": 0.2121320515871048, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9012552499771118, "rewards/repeat_soft/std": 0.09362088888883591, "rewards/judge_quality/mean": 0.2849999964237213, "rewards/judge_quality/std": 0.15390163660049438, "rewards/total_composite/mean": 0.6005358099937439, "rewards/total_composite/std": 0.2662653923034668, "reward": 0.6005358099937439, "reward_std": 0.2662653625011444, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2063148021697998, "sampling/sampling_logp_difference/max": 1.7369320392608643, "sampling/importance_sampling_ratio/min": 0.17605972290039062, "sampling/importance_sampling_ratio/mean": 1.0459977388381958, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5269535183906555, "clip_ratio/low_mean": 0.009765625, "clip_ratio/low_min": 0.009765625, "clip_ratio/high_mean": 0.15762987174093723, "clip_ratio/high_max": 0.15762987174093723, "clip_ratio/region_mean": 0.16739549674093723, "reward_total_mean": 0.6005358099937439, "reward_meter_mean": 0.7598826289176941, "reward_meter_std": 0.40752553939819336, "reward_count_adherence_mean": 0.675000011920929, "reward_count_adherence_std": 0.2121320515871048, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9012552499771118, "reward_repeat_soft_std": 0.09362088888883591, "reward_judge_quality_mean": 0.2849999964237213, "reward_judge_quality_std": 0.15390163660049438, "reward_total_composite_mean": 0.6005358099937439, "reward_total_composite_std": 0.2662653923034668} {"timestamp_utc": "2026-04-13T00:33:59Z", "mode": "train", "global_step": 933, "epoch": 0.09372174786539428, "loss": 0.3865, "grad_norm": 35.79892349243164, "learning_rate": 7.175757575757576e-06, "num_tokens": 1690780.0, "completions/mean_length": 21.5, "completions/min_length": 18.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.5, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.8294287323951721, "rewards/meter/std": 0.27840927243232727, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9660090208053589, "rewards/repeat_soft/std": 0.012209683656692505, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.6460351943969727, "rewards/total_composite/std": 0.28909584879875183, "reward": 0.6460351943969727, "reward_std": 0.28909581899642944, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19873501360416412, "sampling/sampling_logp_difference/max": 1.873253345489502, "sampling/importance_sampling_ratio/min": 0.1536230593919754, "sampling/importance_sampling_ratio/mean": 1.047661542892456, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6704398319125175, "clip_ratio/low_mean": 0.058427318930625916, "clip_ratio/low_min": 0.058427318930625916, "clip_ratio/high_mean": 0.1202485365793109, "clip_ratio/high_max": 0.1202485365793109, "clip_ratio/region_mean": 0.1786758555099368, "reward_total_mean": 0.6460351943969727, "reward_meter_mean": 0.8294287323951721, "reward_meter_std": 0.27840927243232727, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9660090208053589, "reward_repeat_soft_std": 0.012209683656692505, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.6460351943969727, "reward_total_composite_std": 0.28909584879875183} {"timestamp_utc": "2026-04-13T00:34:06Z", "mode": "train", "global_step": 934, "epoch": 0.09382219989954796, "loss": 0.0173, "grad_norm": 20.229787826538086, "learning_rate": 7.172727272727273e-06, "num_tokens": 1692487.0, "completions/mean_length": 39.375, "completions/min_length": 37.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.375, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9679242372512817, "rewards/meter/std": 0.04944360628724098, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9659661650657654, "rewards/repeat_soft/std": 0.03133900463581085, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.8126624822616577, "rewards/total_composite/std": 0.02340312860906124, "reward": 0.8126624822616577, "reward_std": 0.02340313419699669, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17893902957439423, "sampling/sampling_logp_difference/max": 1.3703951835632324, "sampling/importance_sampling_ratio/min": 0.3002624809741974, "sampling/importance_sampling_ratio/mean": 1.0238696336746216, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3724722266197205, "clip_ratio/low_mean": 0.033250355161726475, "clip_ratio/low_min": 0.033250355161726475, "clip_ratio/high_mean": 0.1215713582932949, "clip_ratio/high_max": 0.1215713582932949, "clip_ratio/region_mean": 0.15482171345502138, "reward_total_mean": 0.8126624822616577, "reward_meter_mean": 0.9679242372512817, "reward_meter_std": 0.04944360628724098, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9659661650657654, "reward_repeat_soft_std": 0.03133900463581085, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.8126624822616577, "reward_total_composite_std": 0.02340312860906124} {"timestamp_utc": "2026-04-13T00:34:19Z", "mode": "train", "global_step": 935, "epoch": 0.09392265193370165, "loss": -0.0998, "grad_norm": 2.163970947265625, "learning_rate": 7.16969696969697e-06, "num_tokens": 1694032.0, "completions/mean_length": 91.125, "completions/min_length": 27.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 31.000001907348633, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.7785333395004272, "rewards/meter/std": 0.29398614168167114, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9426136016845703, "rewards/repeat_soft/std": 0.05624711886048317, "rewards/judge_quality/mean": 0.3999999761581421, "rewards/judge_quality/std": 0.1414213478565216, "rewards/total_composite/mean": 0.6723378300666809, "rewards/total_composite/std": 0.28083112835884094, "reward": 0.6723378300666809, "reward_std": 0.28083112835884094, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1252843141555786, "sampling/sampling_logp_difference/max": 0.9494726657867432, "sampling/importance_sampling_ratio/min": 0.38694503903388977, "sampling/importance_sampling_ratio/mean": 1.0141456127166748, "sampling/importance_sampling_ratio/max": 1.7498351335525513, "entropy": 0.7024920508265495, "clip_ratio/low_mean": 0.012096773833036423, "clip_ratio/low_min": 0.012096773833036423, "clip_ratio/high_mean": 0.09485788084566593, "clip_ratio/high_max": 0.09485788084566593, "clip_ratio/region_mean": 0.10695465467870235, "reward_total_mean": 0.6723378300666809, "reward_meter_mean": 0.7785333395004272, "reward_meter_std": 0.29398614168167114, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9426136016845703, "reward_repeat_soft_std": 0.05624711886048317, "reward_judge_quality_mean": 0.3999999761581421, "reward_judge_quality_std": 0.1414213478565216, "reward_total_composite_mean": 0.6723378300666809, "reward_total_composite_std": 0.28083112835884094} {"timestamp_utc": "2026-04-13T00:34:26Z", "mode": "train", "global_step": 936, "epoch": 0.09402310396785535, "loss": 0.0816, "grad_norm": 12.954182624816895, "learning_rate": 7.166666666666667e-06, "num_tokens": 1695847.0, "completions/mean_length": 49.875, "completions/min_length": 35.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.875, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9217723608016968, "rewards/meter/std": 0.19225460290908813, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.11785111576318741, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8726575970649719, "rewards/repeat_soft/std": 0.11804068088531494, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.7088133096694946, "rewards/total_composite/std": 0.10135124623775482, "reward": 0.7088133096694946, "reward_std": 0.10135125368833542, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1314087063074112, "sampling/sampling_logp_difference/max": 2.0259432792663574, "sampling/importance_sampling_ratio/min": 0.1318693906068802, "sampling/importance_sampling_ratio/mean": 1.014020323753357, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9538766369223595, "clip_ratio/low_mean": 0.03780430369079113, "clip_ratio/low_min": 0.03780430369079113, "clip_ratio/high_mean": 0.07725834753364325, "clip_ratio/high_max": 0.07725834753364325, "clip_ratio/region_mean": 0.11506265122443438, "reward_total_mean": 0.7088133096694946, "reward_meter_mean": 0.9217723608016968, "reward_meter_std": 0.19225460290908813, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.11785111576318741, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8726575970649719, "reward_repeat_soft_std": 0.11804068088531494, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.7088133096694946, "reward_total_composite_std": 0.10135124623775482} {"timestamp_utc": "2026-04-13T00:34:33Z", "mode": "train", "global_step": 937, "epoch": 0.09412355600200904, "loss": -0.0247, "grad_norm": 17.287193298339844, "learning_rate": 7.163636363636363e-06, "num_tokens": 1697591.0, "completions/mean_length": 48.0, "completions/min_length": 41.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.0, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.7263710498809814, "rewards/meter/std": 0.2264653891324997, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9588704109191895, "rewards/repeat_soft/std": 0.0329214483499527, "rewards/judge_quality/mean": 0.5575000047683716, "rewards/judge_quality/std": 0.19955310225486755, "rewards/total_composite/mean": 0.70250403881073, "rewards/total_composite/std": 0.14384497702121735, "reward": 0.70250403881073, "reward_std": 0.14384497702121735, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16242121160030365, "sampling/sampling_logp_difference/max": 2.2895705699920654, "sampling/importance_sampling_ratio/min": 0.10130995512008667, "sampling/importance_sampling_ratio/mean": 1.0333805084228516, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0288015380501747, "clip_ratio/low_mean": 0.07900152448564768, "clip_ratio/low_min": 0.07900152448564768, "clip_ratio/high_mean": 0.07025058288127184, "clip_ratio/high_max": 0.07025058288127184, "clip_ratio/region_mean": 0.14925210736691952, "reward_total_mean": 0.70250403881073, "reward_meter_mean": 0.7263710498809814, "reward_meter_std": 0.2264653891324997, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9588704109191895, "reward_repeat_soft_std": 0.0329214483499527, "reward_judge_quality_mean": 0.5575000047683716, "reward_judge_quality_std": 0.19955310225486755, "reward_total_composite_mean": 0.70250403881073, "reward_total_composite_std": 0.14384497702121735} {"timestamp_utc": "2026-04-13T00:34:46Z", "mode": "train", "global_step": 938, "epoch": 0.09422400803616274, "loss": -0.0402, "grad_norm": 2.6655454635620117, "learning_rate": 7.1606060606060615e-06, "num_tokens": 1699632.0, "completions/mean_length": 160.125, "completions/min_length": 59.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 109.85714721679688, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 375.0, "rewards/meter/mean": 0.7652813196182251, "rewards/meter/std": 0.342664510011673, "rewards/count_adherence/mean": 0.5833333730697632, "rewards/count_adherence/std": 0.38832157850265503, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8941712379455566, "rewards/repeat_soft/std": 0.17319457232952118, "rewards/judge_quality/mean": 0.4049999713897705, "rewards/judge_quality/std": 0.2265896201133728, "rewards/total_composite/mean": 0.6427937746047974, "rewards/total_composite/std": 0.23028042912483215, "reward": 0.6427937746047974, "reward_std": 0.23028039932250977, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10395810753107071, "sampling/sampling_logp_difference/max": 1.3975685834884644, "sampling/importance_sampling_ratio/min": 0.24719727039337158, "sampling/importance_sampling_ratio/mean": 1.0160013437271118, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8268853649497032, "clip_ratio/low_mean": 0.025825136341154575, "clip_ratio/low_min": 0.025825136341154575, "clip_ratio/high_mean": 0.07980856951326132, "clip_ratio/high_max": 0.07980856951326132, "clip_ratio/region_mean": 0.1056337058544159, "reward_total_mean": 0.6427937746047974, "reward_meter_mean": 0.7652813196182251, "reward_meter_std": 0.342664510011673, "reward_count_adherence_mean": 0.5833333730697632, "reward_count_adherence_std": 0.38832157850265503, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8941712379455566, "reward_repeat_soft_std": 0.17319457232952118, "reward_judge_quality_mean": 0.4049999713897705, "reward_judge_quality_std": 0.2265896201133728, "reward_total_composite_mean": 0.6427937746047974, "reward_total_composite_std": 0.23028042912483215} {"timestamp_utc": "2026-04-13T00:34:53Z", "mode": "train", "global_step": 939, "epoch": 0.09432446007031642, "loss": 0.0188, "grad_norm": 17.43612289428711, "learning_rate": 7.157575757575758e-06, "num_tokens": 1701225.0, "completions/mean_length": 31.125, "completions/min_length": 28.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.125, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9707468748092651, "rewards/meter/std": 0.03171354904770851, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8886878490447998, "rewards/repeat_soft/std": 0.13332723081111908, "rewards/judge_quality/mean": 0.5024999976158142, "rewards/judge_quality/std": 0.21224987506866455, "rewards/total_composite/mean": 0.8170799016952515, "rewards/total_composite/std": 0.07667647302150726, "reward": 0.8170799016952515, "reward_std": 0.07667648792266846, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1569925993680954, "sampling/sampling_logp_difference/max": 1.5575730800628662, "sampling/importance_sampling_ratio/min": 0.21064667403697968, "sampling/importance_sampling_ratio/mean": 1.0100853443145752, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0956867560744286, "clip_ratio/low_mean": 0.08028402179479599, "clip_ratio/low_min": 0.08028402179479599, "clip_ratio/high_mean": 0.038636364974081516, "clip_ratio/high_max": 0.038636364974081516, "clip_ratio/region_mean": 0.1189203867688775, "reward_total_mean": 0.8170799016952515, "reward_meter_mean": 0.9707468748092651, "reward_meter_std": 0.03171354904770851, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8886878490447998, "reward_repeat_soft_std": 0.13332723081111908, "reward_judge_quality_mean": 0.5024999976158142, "reward_judge_quality_std": 0.21224987506866455, "reward_total_composite_mean": 0.8170799016952515, "reward_total_composite_std": 0.07667647302150726} {"timestamp_utc": "2026-04-13T00:35:01Z", "mode": "train", "global_step": 940, "epoch": 0.09442491210447011, "loss": 0.165, "grad_norm": 28.49531364440918, "learning_rate": 7.154545454545455e-06, "num_tokens": 1702724.0, "completions/mean_length": 36.375, "completions/min_length": 32.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.375, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.42921364307403564, "rewards/meter/std": 0.40255698561668396, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9958227276802063, "rewards/repeat_soft/std": 0.005519228056073189, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.5401498675346375, "rewards/total_composite/std": 0.26159805059432983, "reward": 0.5401498675346375, "reward_std": 0.26159805059432983, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2039574235677719, "sampling/sampling_logp_difference/max": 1.59712553024292, "sampling/importance_sampling_ratio/min": 0.20247770845890045, "sampling/importance_sampling_ratio/mean": 1.0252234935760498, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4508038759231567, "clip_ratio/low_mean": 0.08148938789963722, "clip_ratio/low_min": 0.08148938789963722, "clip_ratio/high_mean": 0.08055603317916393, "clip_ratio/high_max": 0.08055603317916393, "clip_ratio/region_mean": 0.16204542107880116, "reward_total_mean": 0.5401498675346375, "reward_meter_mean": 0.42921364307403564, "reward_meter_std": 0.40255698561668396, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9958227276802063, "reward_repeat_soft_std": 0.005519228056073189, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.5401498675346375, "reward_total_composite_std": 0.26159805059432983} {"timestamp_utc": "2026-04-13T00:35:08Z", "mode": "train", "global_step": 941, "epoch": 0.09452536413862381, "loss": -0.0065, "grad_norm": 11.679455757141113, "learning_rate": 7.151515151515152e-06, "num_tokens": 1704061.0, "completions/mean_length": 31.125, "completions/min_length": 27.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.125, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9915796518325806, "rewards/meter/std": 0.009904453530907631, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9534221887588501, "rewards/repeat_soft/std": 0.01803978905081749, "rewards/judge_quality/mean": 0.3799999952316284, "rewards/judge_quality/std": 0.11501552164554596, "rewards/total_composite/mean": 0.8055530786514282, "rewards/total_composite/std": 0.033388055860996246, "reward": 0.8055530786514282, "reward_std": 0.03338808938860893, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10070651769638062, "sampling/sampling_logp_difference/max": 1.1807641983032227, "sampling/importance_sampling_ratio/min": 0.30704399943351746, "sampling/importance_sampling_ratio/mean": 1.0188117027282715, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6605461947619915, "clip_ratio/low_mean": 0.032258063554763794, "clip_ratio/low_min": 0.032258063554763794, "clip_ratio/high_mean": 0.09027670603245497, "clip_ratio/high_max": 0.09027670603245497, "clip_ratio/region_mean": 0.12253476958721876, "reward_total_mean": 0.8055530786514282, "reward_meter_mean": 0.9915796518325806, "reward_meter_std": 0.009904453530907631, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9534221887588501, "reward_repeat_soft_std": 0.01803978905081749, "reward_judge_quality_mean": 0.3799999952316284, "reward_judge_quality_std": 0.11501552164554596, "reward_total_composite_mean": 0.8055530786514282, "reward_total_composite_std": 0.033388055860996246} {"timestamp_utc": "2026-04-13T00:35:15Z", "mode": "train", "global_step": 942, "epoch": 0.0946258161727775, "loss": 0.0342, "grad_norm": 27.224130630493164, "learning_rate": 7.148484848484849e-06, "num_tokens": 1705519.0, "completions/mean_length": 20.25, "completions/min_length": 19.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.25, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.8797869682312012, "rewards/meter/std": 0.2928362190723419, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.6812499761581421, "rewards/judge_quality/std": 0.25542333722114563, "rewards/total_composite/mean": 0.8465291261672974, "rewards/total_composite/std": 0.12721551954746246, "reward": 0.8465291261672974, "reward_std": 0.12721551954746246, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15355998277664185, "sampling/sampling_logp_difference/max": 2.366159439086914, "sampling/importance_sampling_ratio/min": 0.09384042769670486, "sampling/importance_sampling_ratio/mean": 1.0133529901504517, "sampling/importance_sampling_ratio/max": 1.7101335525512695, "entropy": 0.8487899526953697, "clip_ratio/low_mean": 0.059412799309939146, "clip_ratio/low_min": 0.059412799309939146, "clip_ratio/high_mean": 0.038815789856016636, "clip_ratio/high_max": 0.038815789856016636, "clip_ratio/region_mean": 0.09822858916595578, "reward_total_mean": 0.8465291261672974, "reward_meter_mean": 0.8797869682312012, "reward_meter_std": 0.2928362190723419, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.6812499761581421, "reward_judge_quality_std": 0.25542333722114563, "reward_total_composite_mean": 0.8465291261672974, "reward_total_composite_std": 0.12721551954746246} {"timestamp_utc": "2026-04-13T00:35:24Z", "mode": "train", "global_step": 943, "epoch": 0.0947262682069312, "loss": 0.0378, "grad_norm": 19.396610260009766, "learning_rate": 7.145454545454547e-06, "num_tokens": 1707430.0, "completions/mean_length": 34.875, "completions/min_length": 33.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9316346049308777, "rewards/meter/std": 0.1364348828792572, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9724018573760986, "rewards/repeat_soft/std": 0.04127771407365799, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.7612257599830627, "rewards/total_composite/std": 0.08715901523828506, "reward": 0.7612257599830627, "reward_std": 0.08715900778770447, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16672179102897644, "sampling/sampling_logp_difference/max": 1.638495922088623, "sampling/importance_sampling_ratio/min": 0.1942720264196396, "sampling/importance_sampling_ratio/mean": 1.0350145101547241, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9398013204336166, "clip_ratio/low_mean": 0.05992063647136092, "clip_ratio/low_min": 0.05992063647136092, "clip_ratio/high_mean": 0.07304162532091141, "clip_ratio/high_max": 0.07304162532091141, "clip_ratio/region_mean": 0.13296226179227233, "reward_total_mean": 0.7612257599830627, "reward_meter_mean": 0.9316346049308777, "reward_meter_std": 0.1364348828792572, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9724018573760986, "reward_repeat_soft_std": 0.04127771407365799, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.7612257599830627, "reward_total_composite_std": 0.08715901523828506} {"timestamp_utc": "2026-04-13T00:35:31Z", "mode": "train", "global_step": 944, "epoch": 0.09482672024108488, "loss": 0.0466, "grad_norm": 22.824050903320312, "learning_rate": 7.142424242424243e-06, "num_tokens": 1708792.0, "completions/mean_length": 18.25, "completions/min_length": 16.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 18.25, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.9649356603622437, "rewards/meter/std": 0.07012352347373962, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.8023461103439331, "rewards/total_composite/std": 0.03489043191075325, "reward": 0.8023461103439331, "reward_std": 0.034890417009592056, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16288729012012482, "sampling/sampling_logp_difference/max": 1.4404186010360718, "sampling/importance_sampling_ratio/min": 0.23682859539985657, "sampling/importance_sampling_ratio/mean": 1.0199565887451172, "sampling/importance_sampling_ratio/max": 1.7292900085449219, "entropy": 1.076153539121151, "clip_ratio/low_mean": 0.04102167207747698, "clip_ratio/low_min": 0.04102167207747698, "clip_ratio/high_mean": 0.1464562425389886, "clip_ratio/high_max": 0.1464562425389886, "clip_ratio/region_mean": 0.18747791461646557, "reward_total_mean": 0.8023461103439331, "reward_meter_mean": 0.9649356603622437, "reward_meter_std": 0.07012352347373962, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.8023461103439331, "reward_total_composite_std": 0.03489043191075325} {"timestamp_utc": "2026-04-13T00:35:39Z", "mode": "train", "global_step": 945, "epoch": 0.09492717227523857, "loss": -0.0477, "grad_norm": 11.600089073181152, "learning_rate": 7.1393939393939405e-06, "num_tokens": 1710345.0, "completions/mean_length": 39.125, "completions/min_length": 32.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.125, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.868249773979187, "rewards/meter/std": 0.26648950576782227, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9794807434082031, "rewards/repeat_soft/std": 0.044884320348501205, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.10260014235973358, "rewards/total_composite/mean": 0.7200618982315063, "rewards/total_composite/std": 0.2937321662902832, "reward": 0.7200618982315063, "reward_std": 0.2937321960926056, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1721835881471634, "sampling/sampling_logp_difference/max": 1.4057483673095703, "sampling/importance_sampling_ratio/min": 0.24518351256847382, "sampling/importance_sampling_ratio/mean": 1.0334396362304688, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.403893619775772, "clip_ratio/low_mean": 0.0234375, "clip_ratio/low_min": 0.0234375, "clip_ratio/high_mean": 0.12654181942343712, "clip_ratio/high_max": 0.12654181942343712, "clip_ratio/region_mean": 0.14997931942343712, "reward_total_mean": 0.7200618982315063, "reward_meter_mean": 0.868249773979187, "reward_meter_std": 0.26648950576782227, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9794807434082031, "reward_repeat_soft_std": 0.044884320348501205, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.10260014235973358, "reward_total_composite_mean": 0.7200618982315063, "reward_total_composite_std": 0.2937321662902832} {"timestamp_utc": "2026-04-13T00:35:46Z", "mode": "train", "global_step": 946, "epoch": 0.09502762430939227, "loss": -0.0191, "grad_norm": 12.906012535095215, "learning_rate": 7.136363636363637e-06, "num_tokens": 1712316.0, "completions/mean_length": 47.375, "completions/min_length": 42.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.375, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9802955985069275, "rewards/meter/std": 0.011432122439146042, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7738319635391235, "rewards/repeat_soft/std": 0.1516496241092682, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7442662119865417, "rewards/total_composite/std": 0.034936629235744476, "reward": 0.7442662119865417, "reward_std": 0.03493664041161537, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15090124309062958, "sampling/sampling_logp_difference/max": 1.343179702758789, "sampling/importance_sampling_ratio/min": 0.2610144019126892, "sampling/importance_sampling_ratio/mean": 1.0385605096817017, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2154610380530357, "clip_ratio/low_mean": 0.05039231572300196, "clip_ratio/low_min": 0.05039231572300196, "clip_ratio/high_mean": 0.09790418203920126, "clip_ratio/high_max": 0.09790418203920126, "clip_ratio/region_mean": 0.14829649776220322, "reward_total_mean": 0.7442662119865417, "reward_meter_mean": 0.9802955985069275, "reward_meter_std": 0.011432122439146042, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7738319635391235, "reward_repeat_soft_std": 0.1516496241092682, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7442662119865417, "reward_total_composite_std": 0.034936629235744476} {"timestamp_utc": "2026-04-13T00:35:54Z", "mode": "train", "global_step": 947, "epoch": 0.09512807634354596, "loss": 0.0585, "grad_norm": 17.166635513305664, "learning_rate": 7.133333333333334e-06, "num_tokens": 1714225.0, "completions/mean_length": 54.625, "completions/min_length": 48.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.625, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.787930965423584, "rewards/meter/std": 0.2836931347846985, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9155566692352295, "rewards/repeat_soft/std": 0.08768636733293533, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.6055920124053955, "rewards/total_composite/std": 0.2742660343647003, "reward": 0.6055920124053955, "reward_std": 0.2742660343647003, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18250863254070282, "sampling/sampling_logp_difference/max": 1.659341812133789, "sampling/importance_sampling_ratio/min": 0.19026416540145874, "sampling/importance_sampling_ratio/mean": 1.0212581157684326, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5785379856824875, "clip_ratio/low_mean": 0.030838041566312313, "clip_ratio/low_min": 0.030838041566312313, "clip_ratio/high_mean": 0.1407758817076683, "clip_ratio/high_max": 0.1407758817076683, "clip_ratio/region_mean": 0.17161392327398062, "reward_total_mean": 0.6055920124053955, "reward_meter_mean": 0.787930965423584, "reward_meter_std": 0.2836931347846985, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9155566692352295, "reward_repeat_soft_std": 0.08768636733293533, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.6055920124053955, "reward_total_composite_std": 0.2742660343647003} {"timestamp_utc": "2026-04-13T00:36:05Z", "mode": "train", "global_step": 948, "epoch": 0.09522852837769964, "loss": -0.0442, "grad_norm": 1.909711480140686, "learning_rate": 7.130303030303031e-06, "num_tokens": 1715497.0, "completions/mean_length": 140.0, "completions/min_length": 11.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 16.0, "completions/min_terminated_length": 11.0, "completions/max_terminated_length": 19.0, "rewards/meter/mean": 0.75006103515625, "rewards/meter/std": 0.36651161313056946, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9535620212554932, "rewards/repeat_soft/std": 0.048312265425920486, "rewards/judge_quality/mean": 0.46000000834465027, "rewards/judge_quality/std": 0.3301082253456116, "rewards/total_composite/mean": 0.6003644466400146, "rewards/total_composite/std": 0.3938620090484619, "reward": 0.6003644466400146, "reward_std": 0.3938620090484619, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19879871606826782, "sampling/sampling_logp_difference/max": 1.1840801239013672, "sampling/importance_sampling_ratio/min": 0.3060275614261627, "sampling/importance_sampling_ratio/mean": 1.0619258880615234, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0581244453787804, "clip_ratio/low_mean": 0.02777777798473835, "clip_ratio/low_min": 0.02777777798473835, "clip_ratio/high_mean": 0.09360047802329063, "clip_ratio/high_max": 0.09360047802329063, "clip_ratio/region_mean": 0.12137825600802898, "reward_total_mean": 0.6003644466400146, "reward_meter_mean": 0.75006103515625, "reward_meter_std": 0.36651161313056946, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9535620212554932, "reward_repeat_soft_std": 0.048312265425920486, "reward_judge_quality_mean": 0.46000000834465027, "reward_judge_quality_std": 0.3301082253456116, "reward_total_composite_mean": 0.6003644466400146, "reward_total_composite_std": 0.3938620090484619} {"timestamp_utc": "2026-04-13T00:36:12Z", "mode": "train", "global_step": 949, "epoch": 0.09532898041185334, "loss": 0.0198, "grad_norm": 23.403194427490234, "learning_rate": 7.127272727272728e-06, "num_tokens": 1716823.0, "completions/mean_length": 17.75, "completions/min_length": 15.0, "completions/max_length": 22.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 17.75, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 22.0, "rewards/meter/mean": 0.9446767568588257, "rewards/meter/std": 0.061988186091184616, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9318946599960327, "rewards/repeat_soft/std": 0.08656495064496994, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.263435423374176, "rewards/total_composite/mean": 0.8704190254211426, "rewards/total_composite/std": 0.06929749995470047, "reward": 0.8704190254211426, "reward_std": 0.06929749250411987, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18433409929275513, "sampling/sampling_logp_difference/max": 0.9779407978057861, "sampling/importance_sampling_ratio/min": 0.37608474493026733, "sampling/importance_sampling_ratio/mean": 1.0180706977844238, "sampling/importance_sampling_ratio/max": 1.9708151817321777, "entropy": 1.1104856133460999, "clip_ratio/low_mean": 0.12237506732344627, "clip_ratio/low_min": 0.12237506732344627, "clip_ratio/high_mean": 0.09017565380781889, "clip_ratio/high_max": 0.09017565380781889, "clip_ratio/region_mean": 0.21255072113126516, "reward_total_mean": 0.8704190254211426, "reward_meter_mean": 0.9446767568588257, "reward_meter_std": 0.061988186091184616, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9318946599960327, "reward_repeat_soft_std": 0.08656495064496994, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.263435423374176, "reward_total_composite_mean": 0.8704190254211426, "reward_total_composite_std": 0.06929749995470047} {"timestamp_utc": "2026-04-13T00:36:19Z", "mode": "train", "global_step": 950, "epoch": 0.09542943244600703, "loss": -0.0317, "grad_norm": 16.178081512451172, "learning_rate": 7.124242424242424e-06, "num_tokens": 1718418.0, "completions/mean_length": 35.375, "completions/min_length": 29.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.375, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.9479435682296753, "rewards/meter/std": 0.09432618319988251, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8042763471603394, "rewards/repeat_soft/std": 0.07961636036634445, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7330021858215332, "rewards/total_composite/std": 0.04175588861107826, "reward": 0.7330021858215332, "reward_std": 0.04175589978694916, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14429086446762085, "sampling/sampling_logp_difference/max": 2.102104663848877, "sampling/importance_sampling_ratio/min": 0.12219896912574768, "sampling/importance_sampling_ratio/mean": 1.007461667060852, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7787586525082588, "clip_ratio/low_mean": 0.01293103490024805, "clip_ratio/low_min": 0.01293103490024805, "clip_ratio/high_mean": 0.11816581524908543, "clip_ratio/high_max": 0.11816581524908543, "clip_ratio/region_mean": 0.13109685014933348, "reward_total_mean": 0.7330021858215332, "reward_meter_mean": 0.9479435682296753, "reward_meter_std": 0.09432618319988251, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8042763471603394, "reward_repeat_soft_std": 0.07961636036634445, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7330021858215332, "reward_total_composite_std": 0.04175588861107826} {"timestamp_utc": "2026-04-13T00:37:17Z", "mode": "eval", "global_step": 950, "epoch": 0.09542943244600703, "eval_loss": NaN, "eval_runtime": 58.0428, "eval_samples_per_second": 1.378, "eval_steps_per_second": 0.172, "eval_num_tokens": 1718418.0, "eval_completions/mean_length": 66.0875, "eval_completions/min_length": 27.3, "eval_completions/max_length": 180.4, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 54.69285774230957, "eval_completions/min_terminated_length": 27.3, "eval_completions/max_terminated_length": 101.0, "eval_rewards/meter/mean": 0.8244062244892121, "eval_rewards/meter/std": 0.24955528136342764, "eval_rewards/count_adherence/mean": 0.8406250059604645, "eval_rewards/count_adherence/std": 0.1301371917128563, "eval_rewards/hard_gate/mean": 0.9625, "eval_rewards/hard_gate/std": 0.10606601536273956, "eval_rewards/repeat_soft/mean": 0.8891917705535889, "eval_rewards/repeat_soft/std": 0.11155816726386547, "eval_rewards/judge_quality/mean": 0.4021250009536743, "eval_rewards/judge_quality/std": 0.15732120722532272, "eval_rewards/total_composite/mean": 0.6882369875907898, "eval_rewards/total_composite/std": 0.16960841733962298, "eval_reward": 0.6882369875907898, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.08290928304195404, "eval_sampling/sampling_logp_difference/max": 0.9423806190490722, "eval_sampling/importance_sampling_ratio/min": 0.3952860444784164, "eval_sampling/importance_sampling_ratio/mean": 1.0248507261276245, "eval_sampling/importance_sampling_ratio/max": 1.537277913093567, "eval_entropy": 1.0201783359050751, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6882369875907898, "eval_reward_meter_mean": 0.8244062244892121, "eval_reward_meter_std": 0.24955528136342764, "eval_reward_count_adherence_mean": 0.8406250059604645, "eval_reward_count_adherence_std": 0.1301371917128563, "eval_reward_hard_gate_mean": 0.9625, "eval_reward_hard_gate_std": 0.10606601536273956, "eval_reward_repeat_soft_mean": 0.8891917705535889, "eval_reward_repeat_soft_std": 0.11155816726386547, "eval_reward_judge_quality_mean": 0.4021250009536743, "eval_reward_judge_quality_std": 0.15732120722532272, "eval_reward_total_composite_mean": 0.6882369875907898, "eval_reward_total_composite_std": 0.16960841733962298} {"timestamp_utc": "2026-04-13T00:37:27Z", "mode": "train", "global_step": 951, "epoch": 0.09552988448016073, "loss": 0.091, "grad_norm": 24.940099716186523, "learning_rate": 7.121212121212122e-06, "num_tokens": 1719978.0, "completions/mean_length": 40.0, "completions/min_length": 32.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.8577056527137756, "rewards/meter/std": 0.280638724565506, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.813770592212677, "rewards/repeat_soft/std": 0.15899072587490082, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.7482196092605591, "rewards/total_composite/std": 0.13690289855003357, "reward": 0.7482196092605591, "reward_std": 0.13690291345119476, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1446223258972168, "sampling/sampling_logp_difference/max": 4.568408966064453, "sampling/importance_sampling_ratio/min": 0.010374451987445354, "sampling/importance_sampling_ratio/mean": 0.9993516206741333, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8887327834963799, "clip_ratio/low_mean": 0.03671875037252903, "clip_ratio/low_min": 0.03671875037252903, "clip_ratio/high_mean": 0.09137165732681751, "clip_ratio/high_max": 0.09137165732681751, "clip_ratio/region_mean": 0.12809040769934654, "reward_total_mean": 0.7482196092605591, "reward_meter_mean": 0.8577056527137756, "reward_meter_std": 0.280638724565506, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.813770592212677, "reward_repeat_soft_std": 0.15899072587490082, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.7482196092605591, "reward_total_composite_std": 0.13690289855003357} {"timestamp_utc": "2026-04-13T00:37:34Z", "mode": "train", "global_step": 952, "epoch": 0.09563033651431442, "loss": 0.0349, "grad_norm": 27.545665740966797, "learning_rate": 7.118181818181819e-06, "num_tokens": 1721307.0, "completions/mean_length": 28.125, "completions/min_length": 24.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.125, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.7434507608413696, "rewards/meter/std": 0.387460321187973, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9329497814178467, "rewards/repeat_soft/std": 0.07594505697488785, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.6910978555679321, "rewards/total_composite/std": 0.17103013396263123, "reward": 0.6910978555679321, "reward_std": 0.17103013396263123, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20244066417217255, "sampling/sampling_logp_difference/max": 1.6580257415771484, "sampling/importance_sampling_ratio/min": 0.1905147284269333, "sampling/importance_sampling_ratio/mean": 1.0169609785079956, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.30429607629776, "clip_ratio/low_mean": 0.070761495269835, "clip_ratio/low_min": 0.070761495269835, "clip_ratio/high_mean": 0.13939981162548065, "clip_ratio/high_max": 0.13939981162548065, "clip_ratio/region_mean": 0.21016130689531565, "reward_total_mean": 0.6910978555679321, "reward_meter_mean": 0.7434507608413696, "reward_meter_std": 0.387460321187973, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9329497814178467, "reward_repeat_soft_std": 0.07594505697488785, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.6910978555679321, "reward_total_composite_std": 0.17103013396263123} {"timestamp_utc": "2026-04-13T00:37:41Z", "mode": "train", "global_step": 953, "epoch": 0.0957307885484681, "loss": 0.0209, "grad_norm": 32.228939056396484, "learning_rate": 7.115151515151516e-06, "num_tokens": 1722544.0, "completions/mean_length": 15.625, "completions/min_length": 15.0, "completions/max_length": 16.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 15.625, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 16.0, "rewards/meter/mean": 0.7510910630226135, "rewards/meter/std": 0.43345877528190613, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.715865969657898, "rewards/total_composite/std": 0.1958502233028412, "reward": 0.715865969657898, "reward_std": 0.1958502233028412, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17309825122356415, "sampling/sampling_logp_difference/max": 1.7677087783813477, "sampling/importance_sampling_ratio/min": 0.17072370648384094, "sampling/importance_sampling_ratio/mean": 0.9900714755058289, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.235918827354908, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.11145833414047956, "clip_ratio/high_max": 0.11145833414047956, "clip_ratio/region_mean": 0.11145833414047956, "reward_total_mean": 0.715865969657898, "reward_meter_mean": 0.7510910630226135, "reward_meter_std": 0.43345877528190613, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.715865969657898, "reward_total_composite_std": 0.1958502233028412} {"timestamp_utc": "2026-04-13T00:37:50Z", "mode": "train", "global_step": 954, "epoch": 0.0958312405826218, "loss": 0.0213, "grad_norm": 20.727998733520508, "learning_rate": 7.1121212121212125e-06, "num_tokens": 1723983.0, "completions/mean_length": 25.875, "completions/min_length": 23.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.875, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9235881567001343, "rewards/meter/std": 0.06471528112888336, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9961668252944946, "rewards/repeat_soft/std": 0.0040147071704268456, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.8099813461303711, "rewards/total_composite/std": 0.058487892150878906, "reward": 0.8099813461303711, "reward_std": 0.058487892150878906, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16514702141284943, "sampling/sampling_logp_difference/max": 1.5932753086090088, "sampling/importance_sampling_ratio/min": 0.24814605712890625, "sampling/importance_sampling_ratio/mean": 1.038144588470459, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.118714064359665, "clip_ratio/low_mean": 0.14434147346764803, "clip_ratio/low_min": 0.14434147346764803, "clip_ratio/high_mean": 0.051249999552965164, "clip_ratio/high_max": 0.051249999552965164, "clip_ratio/region_mean": 0.1955914730206132, "reward_total_mean": 0.8099813461303711, "reward_meter_mean": 0.9235881567001343, "reward_meter_std": 0.06471528112888336, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9961668252944946, "reward_repeat_soft_std": 0.0040147071704268456, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.8099813461303711, "reward_total_composite_std": 0.058487892150878906} {"timestamp_utc": "2026-04-13T00:37:58Z", "mode": "train", "global_step": 955, "epoch": 0.09593169261677549, "loss": -0.02, "grad_norm": 9.653182029724121, "learning_rate": 7.10909090909091e-06, "num_tokens": 1726012.0, "completions/mean_length": 71.625, "completions/min_length": 44.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.625, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.618694543838501, "rewards/meter/std": 0.3995482325553894, "rewards/count_adherence/mean": 0.7749999761581421, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8123090863227844, "rewards/repeat_soft/std": 0.0865582525730133, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.5357718467712402, "rewards/total_composite/std": 0.27353599667549133, "reward": 0.5357718467712402, "reward_std": 0.27353599667549133, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14591650664806366, "sampling/sampling_logp_difference/max": 1.5928740501403809, "sampling/importance_sampling_ratio/min": 0.20334036648273468, "sampling/importance_sampling_ratio/mean": 1.019871473312378, "sampling/importance_sampling_ratio/max": 1.9929494857788086, "entropy": 1.3306237012147903, "clip_ratio/low_mean": 0.041141572408378124, "clip_ratio/low_min": 0.041141572408378124, "clip_ratio/high_mean": 0.09251377359032631, "clip_ratio/high_max": 0.09251377359032631, "clip_ratio/region_mean": 0.13365534599870443, "reward_total_mean": 0.5357718467712402, "reward_meter_mean": 0.618694543838501, "reward_meter_std": 0.3995482325553894, "reward_count_adherence_mean": 0.7749999761581421, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8123090863227844, "reward_repeat_soft_std": 0.0865582525730133, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.5357718467712402, "reward_total_composite_std": 0.27353599667549133} {"timestamp_utc": "2026-04-13T00:38:11Z", "mode": "train", "global_step": 956, "epoch": 0.09603214465092919, "loss": -0.1611, "grad_norm": 2.119105577468872, "learning_rate": 7.106060606060606e-06, "num_tokens": 1727892.0, "completions/mean_length": 123.0, "completions/min_length": 58.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 67.42857360839844, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.9263942241668701, "rewards/meter/std": 0.1689123660326004, "rewards/count_adherence/mean": 0.7749999761581421, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7177093029022217, "rewards/repeat_soft/std": 0.1717357188463211, "rewards/judge_quality/mean": 0.3399999737739563, "rewards/judge_quality/std": 0.15052290260791779, "rewards/total_composite/mean": 0.6549023985862732, "rewards/total_composite/std": 0.26680371165275574, "reward": 0.6549023985862732, "reward_std": 0.26680368185043335, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1302904188632965, "sampling/sampling_logp_difference/max": 1.6979246139526367, "sampling/importance_sampling_ratio/min": 0.1830630600452423, "sampling/importance_sampling_ratio/mean": 1.0103206634521484, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9482916668057442, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09401369653642178, "clip_ratio/high_max": 0.09401369653642178, "clip_ratio/region_mean": 0.09401369653642178, "reward_total_mean": 0.6549023985862732, "reward_meter_mean": 0.9263942241668701, "reward_meter_std": 0.1689123660326004, "reward_count_adherence_mean": 0.7749999761581421, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7177093029022217, "reward_repeat_soft_std": 0.1717357188463211, "reward_judge_quality_mean": 0.3399999737739563, "reward_judge_quality_std": 0.15052290260791779, "reward_total_composite_mean": 0.6549023985862732, "reward_total_composite_std": 0.26680371165275574} {"timestamp_utc": "2026-04-13T00:38:18Z", "mode": "train", "global_step": 957, "epoch": 0.09613259668508287, "loss": -0.0473, "grad_norm": 23.69657325744629, "learning_rate": 7.103030303030304e-06, "num_tokens": 1729101.0, "completions/mean_length": 22.125, "completions/min_length": 15.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.125, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.7419511675834656, "rewards/meter/std": 0.3937707543373108, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9622748494148254, "rewards/repeat_soft/std": 0.0006367546156980097, "rewards/judge_quality/mean": 0.32999998331069946, "rewards/judge_quality/std": 0.11747339367866516, "rewards/total_composite/mean": 0.6791055202484131, "rewards/total_composite/std": 0.20116205513477325, "reward": 0.6791055202484131, "reward_std": 0.20116205513477325, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20770807564258575, "sampling/sampling_logp_difference/max": 1.436507225036621, "sampling/importance_sampling_ratio/min": 0.23775674402713776, "sampling/importance_sampling_ratio/mean": 1.0554492473602295, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7637698948383331, "clip_ratio/low_mean": 0.07833333685994148, "clip_ratio/low_min": 0.07833333685994148, "clip_ratio/high_mean": 0.11184860859066248, "clip_ratio/high_max": 0.11184860859066248, "clip_ratio/region_mean": 0.19018194545060396, "reward_total_mean": 0.6791055202484131, "reward_meter_mean": 0.7419511675834656, "reward_meter_std": 0.3937707543373108, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9622748494148254, "reward_repeat_soft_std": 0.0006367546156980097, "reward_judge_quality_mean": 0.32999998331069946, "reward_judge_quality_std": 0.11747339367866516, "reward_total_composite_mean": 0.6791055202484131, "reward_total_composite_std": 0.20116205513477325} {"timestamp_utc": "2026-04-13T00:38:26Z", "mode": "train", "global_step": 958, "epoch": 0.09623304871923656, "loss": -0.0156, "grad_norm": 11.397869110107422, "learning_rate": 7.100000000000001e-06, "num_tokens": 1730751.0, "completions/mean_length": 46.25, "completions/min_length": 26.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.25, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9336819052696228, "rewards/meter/std": 0.1104324534535408, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9718987941741943, "rewards/repeat_soft/std": 0.04406929388642311, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.1970088928937912, "rewards/total_composite/mean": 0.807971715927124, "rewards/total_composite/std": 0.05070125311613083, "reward": 0.807971715927124, "reward_std": 0.05070125311613083, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13506978750228882, "sampling/sampling_logp_difference/max": 1.4036264419555664, "sampling/importance_sampling_ratio/min": 0.24570432305335999, "sampling/importance_sampling_ratio/mean": 0.9950107932090759, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9007581993937492, "clip_ratio/low_mean": 0.07162188272923231, "clip_ratio/low_min": 0.07162188272923231, "clip_ratio/high_mean": 0.06933447998017073, "clip_ratio/high_max": 0.06933447998017073, "clip_ratio/region_mean": 0.14095636270940304, "reward_total_mean": 0.807971715927124, "reward_meter_mean": 0.9336819052696228, "reward_meter_std": 0.1104324534535408, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9718987941741943, "reward_repeat_soft_std": 0.04406929388642311, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.1970088928937912, "reward_total_composite_mean": 0.807971715927124, "reward_total_composite_std": 0.05070125311613083} {"timestamp_utc": "2026-04-13T00:38:32Z", "mode": "train", "global_step": 959, "epoch": 0.09633350075339026, "loss": -0.0564, "grad_norm": 20.887874603271484, "learning_rate": 7.096969696969698e-06, "num_tokens": 1732102.0, "completions/mean_length": 19.875, "completions/min_length": 15.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.875, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.644660234451294, "rewards/meter/std": 0.4049033522605896, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.48875001072883606, "rewards/judge_quality/std": 0.29536598920822144, "rewards/total_composite/mean": 0.6829720735549927, "rewards/total_composite/std": 0.19990114867687225, "reward": 0.6829720735549927, "reward_std": 0.19990113377571106, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1370769441127777, "sampling/sampling_logp_difference/max": 0.9573252201080322, "sampling/importance_sampling_ratio/min": 0.383918434381485, "sampling/importance_sampling_ratio/mean": 1.0443263053894043, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2619379088282585, "clip_ratio/low_mean": 0.068014707416296, "clip_ratio/low_min": 0.068014707416296, "clip_ratio/high_mean": 0.09556393884122372, "clip_ratio/high_max": 0.09556393884122372, "clip_ratio/region_mean": 0.16357864625751972, "reward_total_mean": 0.6829720735549927, "reward_meter_mean": 0.644660234451294, "reward_meter_std": 0.4049033522605896, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.48875001072883606, "reward_judge_quality_std": 0.29536598920822144, "reward_total_composite_mean": 0.6829720735549927, "reward_total_composite_std": 0.19990114867687225} {"timestamp_utc": "2026-04-13T00:38:45Z", "mode": "train", "global_step": 960, "epoch": 0.09643395278754395, "loss": -0.0766, "grad_norm": 3.8090667724609375, "learning_rate": 7.093939393939394e-06, "num_tokens": 1733672.0, "completions/mean_length": 95.25, "completions/min_length": 30.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 35.71428680419922, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.45697295665740967, "rewards/meter/std": 0.4563824236392975, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.952043890953064, "rewards/repeat_soft/std": 0.05059613659977913, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.13845446705818176, "rewards/total_composite/mean": 0.5245922207832336, "rewards/total_composite/std": 0.2877596914768219, "reward": 0.5245922207832336, "reward_std": 0.2877596914768219, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16219273209571838, "sampling/sampling_logp_difference/max": 1.607956886291504, "sampling/importance_sampling_ratio/min": 0.20029643177986145, "sampling/importance_sampling_ratio/mean": 1.0401145219802856, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2723911702632904, "clip_ratio/low_mean": 0.06463675387203693, "clip_ratio/low_min": 0.06463675387203693, "clip_ratio/high_mean": 0.08372637815773487, "clip_ratio/high_max": 0.08372637815773487, "clip_ratio/region_mean": 0.1483631320297718, "reward_total_mean": 0.5245922207832336, "reward_meter_mean": 0.45697295665740967, "reward_meter_std": 0.4563824236392975, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.952043890953064, "reward_repeat_soft_std": 0.05059613659977913, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.13845446705818176, "reward_total_composite_mean": 0.5245922207832336, "reward_total_composite_std": 0.2877596914768219} {"timestamp_utc": "2026-04-13T00:38:52Z", "mode": "train", "global_step": 961, "epoch": 0.09653440482169764, "loss": 0.0299, "grad_norm": 13.221223831176758, "learning_rate": 7.0909090909090916e-06, "num_tokens": 1735569.0, "completions/mean_length": 55.125, "completions/min_length": 46.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.125, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.7280842065811157, "rewards/meter/std": 0.3960942327976227, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8426153063774109, "rewards/repeat_soft/std": 0.1254083216190338, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.1249857097864151, "rewards/total_composite/mean": 0.6676493883132935, "rewards/total_composite/std": 0.1980828195810318, "reward": 0.6676493883132935, "reward_std": 0.1980828046798706, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14404921233654022, "sampling/sampling_logp_difference/max": 2.2121829986572266, "sampling/importance_sampling_ratio/min": 0.10946144163608551, "sampling/importance_sampling_ratio/mean": 0.9917121529579163, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9216381534934044, "clip_ratio/low_mean": 0.031444099731743336, "clip_ratio/low_min": 0.031444099731743336, "clip_ratio/high_mean": 0.07886816235259175, "clip_ratio/high_max": 0.07886816235259175, "clip_ratio/region_mean": 0.11031226208433509, "reward_total_mean": 0.6676493883132935, "reward_meter_mean": 0.7280842065811157, "reward_meter_std": 0.3960942327976227, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8426153063774109, "reward_repeat_soft_std": 0.1254083216190338, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.1249857097864151, "reward_total_composite_mean": 0.6676493883132935, "reward_total_composite_std": 0.1980828195810318} {"timestamp_utc": "2026-04-13T00:38:59Z", "mode": "train", "global_step": 962, "epoch": 0.09663485685585133, "loss": 0.01, "grad_norm": 13.003908157348633, "learning_rate": 7.087878787878788e-06, "num_tokens": 1737051.0, "completions/mean_length": 32.25, "completions/min_length": 30.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.25, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.835488498210907, "rewards/meter/std": 0.3177703320980072, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9835466742515564, "rewards/repeat_soft/std": 0.013539280742406845, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.7990745306015015, "rewards/total_composite/std": 0.12338896840810776, "reward": 0.7990745306015015, "reward_std": 0.12338896840810776, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15376131236553192, "sampling/sampling_logp_difference/max": 1.4138660430908203, "sampling/importance_sampling_ratio/min": 0.24320124089717865, "sampling/importance_sampling_ratio/mean": 1.021579623222351, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1979876011610031, "clip_ratio/low_mean": 0.052343751303851604, "clip_ratio/low_min": 0.052343751303851604, "clip_ratio/high_mean": 0.11093086190521717, "clip_ratio/high_max": 0.11093086190521717, "clip_ratio/region_mean": 0.16327461320906878, "reward_total_mean": 0.7990745306015015, "reward_meter_mean": 0.835488498210907, "reward_meter_std": 0.3177703320980072, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9835466742515564, "reward_repeat_soft_std": 0.013539280742406845, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.7990745306015015, "reward_total_composite_std": 0.12338896840810776} {"timestamp_utc": "2026-04-13T00:39:07Z", "mode": "train", "global_step": 963, "epoch": 0.09673530889000502, "loss": -0.0216, "grad_norm": 14.177504539489746, "learning_rate": 7.084848484848485e-06, "num_tokens": 1738820.0, "completions/mean_length": 44.125, "completions/min_length": 35.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.125, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.8435533046722412, "rewards/meter/std": 0.301093727350235, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9704599976539612, "rewards/repeat_soft/std": 0.03459371253848076, "rewards/judge_quality/mean": 0.41874998807907104, "rewards/judge_quality/std": 0.14574319124221802, "rewards/total_composite/mean": 0.752269983291626, "rewards/total_composite/std": 0.14172376692295074, "reward": 0.752269983291626, "reward_std": 0.14172375202178955, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15943895280361176, "sampling/sampling_logp_difference/max": 1.0679357051849365, "sampling/importance_sampling_ratio/min": 0.3437173366546631, "sampling/importance_sampling_ratio/mean": 1.0318710803985596, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2702634707093239, "clip_ratio/low_mean": 0.044387755915522575, "clip_ratio/low_min": 0.044387755915522575, "clip_ratio/high_mean": 0.1020165616646409, "clip_ratio/high_max": 0.1020165616646409, "clip_ratio/region_mean": 0.14640431758016348, "reward_total_mean": 0.752269983291626, "reward_meter_mean": 0.8435533046722412, "reward_meter_std": 0.301093727350235, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9704599976539612, "reward_repeat_soft_std": 0.03459371253848076, "reward_judge_quality_mean": 0.41874998807907104, "reward_judge_quality_std": 0.14574319124221802, "reward_total_composite_mean": 0.752269983291626, "reward_total_composite_std": 0.14172376692295074} {"timestamp_utc": "2026-04-13T00:39:15Z", "mode": "train", "global_step": 964, "epoch": 0.09683576092415871, "loss": 0.0107, "grad_norm": 22.403032302856445, "learning_rate": 7.081818181818182e-06, "num_tokens": 1740400.0, "completions/mean_length": 33.5, "completions/min_length": 29.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.5, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.7350749373435974, "rewards/meter/std": 0.3701496720314026, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9334726333618164, "rewards/repeat_soft/std": 0.08218740671873093, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.6948809623718262, "rewards/total_composite/std": 0.16182801127433777, "reward": 0.6948809623718262, "reward_std": 0.16182799637317657, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1654091775417328, "sampling/sampling_logp_difference/max": 1.9079742431640625, "sampling/importance_sampling_ratio/min": 0.14838066697120667, "sampling/importance_sampling_ratio/mean": 1.0214755535125732, "sampling/importance_sampling_ratio/max": 1.9421496391296387, "entropy": 1.2006257101893425, "clip_ratio/low_mean": 0.04111731005832553, "clip_ratio/low_min": 0.04111731005832553, "clip_ratio/high_mean": 0.09086117637343705, "clip_ratio/high_max": 0.09086117637343705, "clip_ratio/region_mean": 0.13197848643176258, "reward_total_mean": 0.6948809623718262, "reward_meter_mean": 0.7350749373435974, "reward_meter_std": 0.3701496720314026, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9334726333618164, "reward_repeat_soft_std": 0.08218740671873093, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.6948809623718262, "reward_total_composite_std": 0.16182801127433777} {"timestamp_utc": "2026-04-13T00:39:27Z", "mode": "train", "global_step": 965, "epoch": 0.09693621295831241, "loss": -0.1062, "grad_norm": 3.479693651199341, "learning_rate": 7.07878787878788e-06, "num_tokens": 1741872.0, "completions/mean_length": 95.0, "completions/min_length": 28.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 35.42857360839844, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.7207809090614319, "rewards/meter/std": 0.3848625123500824, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9913332462310791, "rewards/repeat_soft/std": 0.013921435922384262, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.13274572789669037, "rewards/total_composite/mean": 0.6540572643280029, "rewards/total_composite/std": 0.287596195936203, "reward": 0.6540572643280029, "reward_std": 0.287596195936203, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18227238953113556, "sampling/sampling_logp_difference/max": 1.6118078231811523, "sampling/importance_sampling_ratio/min": 0.1995265781879425, "sampling/importance_sampling_ratio/mean": 1.00761079788208, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.069210261106491, "clip_ratio/low_mean": 0.0223214291036129, "clip_ratio/low_min": 0.0223214291036129, "clip_ratio/high_mean": 0.12655992154031992, "clip_ratio/high_max": 0.12655992154031992, "clip_ratio/region_mean": 0.14888135064393282, "reward_total_mean": 0.6540572643280029, "reward_meter_mean": 0.7207809090614319, "reward_meter_std": 0.3848625123500824, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9913332462310791, "reward_repeat_soft_std": 0.013921435922384262, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.13274572789669037, "reward_total_composite_mean": 0.6540572643280029, "reward_total_composite_std": 0.287596195936203} {"timestamp_utc": "2026-04-13T00:39:34Z", "mode": "train", "global_step": 966, "epoch": 0.0970366649924661, "loss": 0.0821, "grad_norm": 22.029136657714844, "learning_rate": 7.075757575757576e-06, "num_tokens": 1743322.0, "completions/mean_length": 24.25, "completions/min_length": 20.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.25, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9407020807266235, "rewards/meter/std": 0.03135357052087784, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.896962583065033, "rewards/repeat_soft/std": 0.058386076241731644, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7912622094154358, "rewards/total_composite/std": 0.019059820100665092, "reward": 0.7912622094154358, "reward_std": 0.01905982382595539, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1304568201303482, "sampling/sampling_logp_difference/max": 1.3099141120910645, "sampling/importance_sampling_ratio/min": 0.2698432505130768, "sampling/importance_sampling_ratio/mean": 1.014512300491333, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7357181757688522, "clip_ratio/low_mean": 0.057043652050197124, "clip_ratio/low_min": 0.057043652050197124, "clip_ratio/high_mean": 0.1114342212677002, "clip_ratio/high_max": 0.1114342212677002, "clip_ratio/region_mean": 0.16847787331789732, "reward_total_mean": 0.7912622094154358, "reward_meter_mean": 0.9407020807266235, "reward_meter_std": 0.03135357052087784, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.896962583065033, "reward_repeat_soft_std": 0.058386076241731644, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7912622094154358, "reward_total_composite_std": 0.019059820100665092} {"timestamp_utc": "2026-04-13T00:39:41Z", "mode": "train", "global_step": 967, "epoch": 0.09713711702661978, "loss": 0.029, "grad_norm": 20.158573150634766, "learning_rate": 7.072727272727273e-06, "num_tokens": 1745205.0, "completions/mean_length": 61.375, "completions/min_length": 50.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.375, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.19420486688613892, "rewards/meter/std": 0.10886930674314499, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9447486400604248, "rewards/repeat_soft/std": 0.05627667158842087, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.45317956805229187, "rewards/total_composite/std": 0.05601803585886955, "reward": 0.45317956805229187, "reward_std": 0.05601804330945015, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13372734189033508, "sampling/sampling_logp_difference/max": 1.4077072143554688, "sampling/importance_sampling_ratio/min": 0.24470369517803192, "sampling/importance_sampling_ratio/mean": 1.0296605825424194, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7752871327102184, "clip_ratio/low_mean": 0.07437185943126678, "clip_ratio/low_min": 0.07437185943126678, "clip_ratio/high_mean": 0.045053225476294756, "clip_ratio/high_max": 0.045053225476294756, "clip_ratio/region_mean": 0.11942508490756154, "reward_total_mean": 0.45317956805229187, "reward_meter_mean": 0.19420486688613892, "reward_meter_std": 0.10886930674314499, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9447486400604248, "reward_repeat_soft_std": 0.05627667158842087, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.45317956805229187, "reward_total_composite_std": 0.05601803585886955} {"timestamp_utc": "2026-04-13T00:39:54Z", "mode": "train", "global_step": 968, "epoch": 0.09723756906077348, "loss": -0.123, "grad_norm": 2.016352891921997, "learning_rate": 7.06969696969697e-06, "num_tokens": 1746807.0, "completions/mean_length": 99.25, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 40.28571701049805, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.8281148076057434, "rewards/meter/std": 0.3247700333595276, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9779752492904663, "rewards/repeat_soft/std": 0.022683070972561836, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.13617216050624847, "rewards/total_composite/mean": 0.6998715400695801, "rewards/total_composite/std": 0.2856396436691284, "reward": 0.6998715400695801, "reward_std": 0.2856396436691284, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16110891103744507, "sampling/sampling_logp_difference/max": 1.7866315841674805, "sampling/importance_sampling_ratio/min": 0.16752350330352783, "sampling/importance_sampling_ratio/mean": 1.0341721773147583, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.047916904091835, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.11845944076776505, "clip_ratio/high_max": 0.11845944076776505, "clip_ratio/region_mean": 0.11845944076776505, "reward_total_mean": 0.6998715400695801, "reward_meter_mean": 0.8281148076057434, "reward_meter_std": 0.3247700333595276, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9779752492904663, "reward_repeat_soft_std": 0.022683070972561836, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.13617216050624847, "reward_total_composite_mean": 0.6998715400695801, "reward_total_composite_std": 0.2856396436691284} {"timestamp_utc": "2026-04-13T00:40:01Z", "mode": "train", "global_step": 969, "epoch": 0.09733802109492717, "loss": -0.0126, "grad_norm": 17.267160415649414, "learning_rate": 7.066666666666667e-06, "num_tokens": 1748292.0, "completions/mean_length": 35.625, "completions/min_length": 29.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.625, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.8851034641265869, "rewards/meter/std": 0.2826894521713257, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9147630929946899, "rewards/repeat_soft/std": 0.06948169320821762, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.19255799055099487, "rewards/total_composite/mean": 0.7815228700637817, "rewards/total_composite/std": 0.14388050138950348, "reward": 0.7815228700637817, "reward_std": 0.14388048648834229, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18947722017765045, "sampling/sampling_logp_difference/max": 1.5001943111419678, "sampling/importance_sampling_ratio/min": 0.22308681905269623, "sampling/importance_sampling_ratio/mean": 1.030003309249878, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0511269941926003, "clip_ratio/low_mean": 0.027058573439717293, "clip_ratio/low_min": 0.027058573439717293, "clip_ratio/high_mean": 0.12381153646856546, "clip_ratio/high_max": 0.12381153646856546, "clip_ratio/region_mean": 0.15087010990828276, "reward_total_mean": 0.7815228700637817, "reward_meter_mean": 0.8851034641265869, "reward_meter_std": 0.2826894521713257, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9147630929946899, "reward_repeat_soft_std": 0.06948169320821762, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.19255799055099487, "reward_total_composite_mean": 0.7815228700637817, "reward_total_composite_std": 0.14388050138950348} {"timestamp_utc": "2026-04-13T00:40:09Z", "mode": "train", "global_step": 970, "epoch": 0.09743847312908087, "loss": 0.0648, "grad_norm": 10.16097354888916, "learning_rate": 7.063636363636365e-06, "num_tokens": 1750076.0, "completions/mean_length": 62.0, "completions/min_length": 57.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9662516117095947, "rewards/meter/std": 0.07716600596904755, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9146057963371277, "rewards/repeat_soft/std": 0.09043062478303909, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.1940544992685318, "rewards/total_composite/mean": 0.8157738447189331, "rewards/total_composite/std": 0.07130660861730576, "reward": 0.8157738447189331, "reward_std": 0.07130659371614456, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09712520241737366, "sampling/sampling_logp_difference/max": 1.122903823852539, "sampling/importance_sampling_ratio/min": 0.32533368468284607, "sampling/importance_sampling_ratio/mean": 1.0112298727035522, "sampling/importance_sampling_ratio/max": 1.9398273229599, "entropy": 0.6823331639170647, "clip_ratio/low_mean": 0.04894965048879385, "clip_ratio/low_min": 0.04894965048879385, "clip_ratio/high_mean": 0.04677743185311556, "clip_ratio/high_max": 0.04677743185311556, "clip_ratio/region_mean": 0.09572708234190941, "reward_total_mean": 0.8157738447189331, "reward_meter_mean": 0.9662516117095947, "reward_meter_std": 0.07716600596904755, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9146057963371277, "reward_repeat_soft_std": 0.09043062478303909, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.1940544992685318, "reward_total_composite_mean": 0.8157738447189331, "reward_total_composite_std": 0.07130660861730576} {"timestamp_utc": "2026-04-13T00:40:20Z", "mode": "train", "global_step": 971, "epoch": 0.09753892516323455, "loss": -0.037, "grad_norm": 2.240598440170288, "learning_rate": 7.060606060606061e-06, "num_tokens": 1751377.0, "completions/mean_length": 144.625, "completions/min_length": 17.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 22.166667938232422, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.37370622158050537, "rewards/meter/std": 0.46938273310661316, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9145724773406982, "rewards/repeat_soft/std": 0.11238186061382294, "rewards/judge_quality/mean": 0.5037499666213989, "rewards/judge_quality/std": 0.4460921883583069, "rewards/total_composite/mean": 0.4944530725479126, "rewards/total_composite/std": 0.4085073471069336, "reward": 0.4944530725479126, "reward_std": 0.4085073471069336, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17249049246311188, "sampling/sampling_logp_difference/max": 1.2068357467651367, "sampling/importance_sampling_ratio/min": 0.29914233088493347, "sampling/importance_sampling_ratio/mean": 1.0064994096755981, "sampling/importance_sampling_ratio/max": 1.730682373046875, "entropy": 1.2030795589089394, "clip_ratio/low_mean": 0.04583333432674408, "clip_ratio/low_min": 0.04583333432674408, "clip_ratio/high_mean": 0.04932010266929865, "clip_ratio/high_max": 0.04932010266929865, "clip_ratio/region_mean": 0.09515343699604273, "reward_total_mean": 0.4944530725479126, "reward_meter_mean": 0.37370622158050537, "reward_meter_std": 0.46938273310661316, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9145724773406982, "reward_repeat_soft_std": 0.11238186061382294, "reward_judge_quality_mean": 0.5037499666213989, "reward_judge_quality_std": 0.4460921883583069, "reward_total_composite_mean": 0.4944530725479126, "reward_total_composite_std": 0.4085073471069336} {"timestamp_utc": "2026-04-13T00:40:28Z", "mode": "train", "global_step": 972, "epoch": 0.09763937719738824, "loss": 0.0756, "grad_norm": 18.723772048950195, "learning_rate": 7.057575757575759e-06, "num_tokens": 1753128.0, "completions/mean_length": 45.875, "completions/min_length": 33.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.850835919380188, "rewards/meter/std": 0.23726779222488403, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.17817415297031403, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9027864336967468, "rewards/repeat_soft/std": 0.1210128664970398, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.6171202659606934, "rewards/total_composite/std": 0.2740100920200348, "reward": 0.6171202659606934, "reward_std": 0.2740100920200348, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17168816924095154, "sampling/sampling_logp_difference/max": 2.2117574214935303, "sampling/importance_sampling_ratio/min": 0.10950803011655807, "sampling/importance_sampling_ratio/mean": 1.0097870826721191, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2804264798760414, "clip_ratio/low_mean": 0.04636330530047417, "clip_ratio/low_min": 0.04636330530047417, "clip_ratio/high_mean": 0.08903860580176115, "clip_ratio/high_max": 0.08903860580176115, "clip_ratio/region_mean": 0.13540191110223532, "reward_total_mean": 0.6171202659606934, "reward_meter_mean": 0.850835919380188, "reward_meter_std": 0.23726779222488403, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.17817415297031403, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9027864336967468, "reward_repeat_soft_std": 0.1210128664970398, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.6171202659606934, "reward_total_composite_std": 0.2740100920200348} {"timestamp_utc": "2026-04-13T00:40:36Z", "mode": "train", "global_step": 973, "epoch": 0.09773982923154194, "loss": 0.114, "grad_norm": 8.647068977355957, "learning_rate": 7.054545454545455e-06, "num_tokens": 1754949.0, "completions/mean_length": 62.625, "completions/min_length": 46.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.625, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9870104789733887, "rewards/meter/std": 0.016873100772500038, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8264553546905518, "rewards/repeat_soft/std": 0.13162322342395782, "rewards/judge_quality/mean": 0.25874999165534973, "rewards/judge_quality/std": 0.11849502474069595, "rewards/total_composite/mean": 0.7544252872467041, "rewards/total_composite/std": 0.04565577208995819, "reward": 0.7544252872467041, "reward_std": 0.04565577208995819, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1112964078783989, "sampling/sampling_logp_difference/max": 1.9966931343078613, "sampling/importance_sampling_ratio/min": 0.13578356802463531, "sampling/importance_sampling_ratio/mean": 1.0102442502975464, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7874184139072895, "clip_ratio/low_mean": 0.04919360298663378, "clip_ratio/low_min": 0.04919360298663378, "clip_ratio/high_mean": 0.0525622358545661, "clip_ratio/high_max": 0.0525622358545661, "clip_ratio/region_mean": 0.10175583884119987, "reward_total_mean": 0.7544252872467041, "reward_meter_mean": 0.9870104789733887, "reward_meter_std": 0.016873100772500038, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8264553546905518, "reward_repeat_soft_std": 0.13162322342395782, "reward_judge_quality_mean": 0.25874999165534973, "reward_judge_quality_std": 0.11849502474069595, "reward_total_composite_mean": 0.7544252872467041, "reward_total_composite_std": 0.04565577208995819} {"timestamp_utc": "2026-04-13T00:40:44Z", "mode": "train", "global_step": 974, "epoch": 0.09784028126569563, "loss": 0.0865, "grad_norm": 12.688441276550293, "learning_rate": 7.0515151515151525e-06, "num_tokens": 1756513.0, "completions/mean_length": 43.5, "completions/min_length": 38.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.5, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.6946251392364502, "rewards/meter/std": 0.42172375321388245, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9058836698532104, "rewards/repeat_soft/std": 0.11435195803642273, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.6904196739196777, "rewards/total_composite/std": 0.19633075594902039, "reward": 0.6904196739196777, "reward_std": 0.19633075594902039, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10494177043437958, "sampling/sampling_logp_difference/max": 1.2281970977783203, "sampling/importance_sampling_ratio/min": 0.2928200364112854, "sampling/importance_sampling_ratio/mean": 1.0248204469680786, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8902135342359543, "clip_ratio/low_mean": 0.039367612451314926, "clip_ratio/low_min": 0.039367612451314926, "clip_ratio/high_mean": 0.06392382085323334, "clip_ratio/high_max": 0.06392382085323334, "clip_ratio/region_mean": 0.10329143330454826, "reward_total_mean": 0.6904196739196777, "reward_meter_mean": 0.6946251392364502, "reward_meter_std": 0.42172375321388245, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9058836698532104, "reward_repeat_soft_std": 0.11435195803642273, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.6904196739196777, "reward_total_composite_std": 0.19633075594902039} {"timestamp_utc": "2026-04-13T00:40:51Z", "mode": "train", "global_step": 975, "epoch": 0.09794073329984933, "loss": -0.0838, "grad_norm": 9.632392883300781, "learning_rate": 7.048484848484849e-06, "num_tokens": 1758281.0, "completions/mean_length": 53.0, "completions/min_length": 38.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.0, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9868560433387756, "rewards/meter/std": 0.01891789212822914, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8803240060806274, "rewards/repeat_soft/std": 0.10478363931179047, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7637426257133484, "rewards/total_composite/std": 0.014769034460186958, "reward": 0.7637426257133484, "reward_std": 0.01476903073489666, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15012013912200928, "sampling/sampling_logp_difference/max": 1.6572530269622803, "sampling/importance_sampling_ratio/min": 0.19066201150417328, "sampling/importance_sampling_ratio/mean": 1.012442946434021, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0591792911291122, "clip_ratio/low_mean": 0.04229068662971258, "clip_ratio/low_min": 0.04229068662971258, "clip_ratio/high_mean": 0.09010986052453518, "clip_ratio/high_max": 0.09010986052453518, "clip_ratio/region_mean": 0.13240054715424776, "reward_total_mean": 0.7637426257133484, "reward_meter_mean": 0.9868560433387756, "reward_meter_std": 0.01891789212822914, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8803240060806274, "reward_repeat_soft_std": 0.10478363931179047, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7637426257133484, "reward_total_composite_std": 0.014769034460186958} {"timestamp_utc": "2026-04-13T00:40:57Z", "mode": "train", "global_step": 976, "epoch": 0.09804118533400301, "loss": -0.0315, "grad_norm": 12.571815490722656, "learning_rate": 7.045454545454546e-06, "num_tokens": 1759840.0, "completions/mean_length": 40.875, "completions/min_length": 29.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.875, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.7463455200195312, "rewards/meter/std": 0.41816213726997375, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9970208406448364, "rewards/repeat_soft/std": 0.005226857960224152, "rewards/judge_quality/mean": 0.5862500071525574, "rewards/judge_quality/std": 0.2822834253311157, "rewards/total_composite/mean": 0.7614325284957886, "rewards/total_composite/std": 0.23450984060764313, "reward": 0.7614325284957886, "reward_std": 0.23450982570648193, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17553706467151642, "sampling/sampling_logp_difference/max": 1.5828590393066406, "sampling/importance_sampling_ratio/min": 0.20538705587387085, "sampling/importance_sampling_ratio/mean": 1.0220659971237183, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2519951462745667, "clip_ratio/low_mean": 0.05100574716925621, "clip_ratio/low_min": 0.05100574716925621, "clip_ratio/high_mean": 0.13577604945749044, "clip_ratio/high_max": 0.13577604945749044, "clip_ratio/region_mean": 0.18678179662674665, "reward_total_mean": 0.7614325284957886, "reward_meter_mean": 0.7463455200195312, "reward_meter_std": 0.41816213726997375, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9970208406448364, "reward_repeat_soft_std": 0.005226857960224152, "reward_judge_quality_mean": 0.5862500071525574, "reward_judge_quality_std": 0.2822834253311157, "reward_total_composite_mean": 0.7614325284957886, "reward_total_composite_std": 0.23450984060764313} {"timestamp_utc": "2026-04-13T00:41:04Z", "mode": "train", "global_step": 977, "epoch": 0.0981416373681567, "loss": 0.0155, "grad_norm": 16.506267547607422, "learning_rate": 7.0424242424242426e-06, "num_tokens": 1761227.0, "completions/mean_length": 23.375, "completions/min_length": 21.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.375, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.7944992780685425, "rewards/meter/std": 0.25980621576309204, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9372137784957886, "rewards/repeat_soft/std": 0.04620867967605591, "rewards/judge_quality/mean": 0.7074999809265137, "rewards/judge_quality/std": 0.2474873960018158, "rewards/total_composite/mean": 0.8134961128234863, "rewards/total_composite/std": 0.13059885799884796, "reward": 0.8134961128234863, "reward_std": 0.13059885799884796, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13714510202407837, "sampling/sampling_logp_difference/max": 1.791841983795166, "sampling/importance_sampling_ratio/min": 0.16665291786193848, "sampling/importance_sampling_ratio/mean": 0.9720542430877686, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.706818088889122, "clip_ratio/low_mean": 0.03714514710009098, "clip_ratio/low_min": 0.03714514710009098, "clip_ratio/high_mean": 0.06306465482339263, "clip_ratio/high_max": 0.06306465482339263, "clip_ratio/region_mean": 0.10020980192348361, "reward_total_mean": 0.8134961128234863, "reward_meter_mean": 0.7944992780685425, "reward_meter_std": 0.25980621576309204, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9372137784957886, "reward_repeat_soft_std": 0.04620867967605591, "reward_judge_quality_mean": 0.7074999809265137, "reward_judge_quality_std": 0.2474873960018158, "reward_total_composite_mean": 0.8134961128234863, "reward_total_composite_std": 0.13059885799884796} {"timestamp_utc": "2026-04-13T00:41:12Z", "mode": "train", "global_step": 978, "epoch": 0.0982420894023104, "loss": -0.0248, "grad_norm": 19.7985897064209, "learning_rate": 7.039393939393941e-06, "num_tokens": 1762942.0, "completions/mean_length": 39.375, "completions/min_length": 34.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.375, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.8185433149337769, "rewards/meter/std": 0.3218686580657959, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9716973304748535, "rewards/repeat_soft/std": 0.020131055265665054, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.08166787773370743, "rewards/total_composite/mean": 0.7298892736434937, "rewards/total_composite/std": 0.14072683453559875, "reward": 0.7298892736434937, "reward_std": 0.14072683453559875, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18384364247322083, "sampling/sampling_logp_difference/max": 1.5088462829589844, "sampling/importance_sampling_ratio/min": 0.2211650013923645, "sampling/importance_sampling_ratio/mean": 0.99518221616745, "sampling/importance_sampling_ratio/max": 1.8186771869659424, "entropy": 1.3427948877215385, "clip_ratio/low_mean": 0.0433441570494324, "clip_ratio/low_min": 0.0433441570494324, "clip_ratio/high_mean": 0.13067620620131493, "clip_ratio/high_max": 0.13067620620131493, "clip_ratio/region_mean": 0.17402036325074732, "reward_total_mean": 0.7298892736434937, "reward_meter_mean": 0.8185433149337769, "reward_meter_std": 0.3218686580657959, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9716973304748535, "reward_repeat_soft_std": 0.020131055265665054, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.08166787773370743, "reward_total_composite_mean": 0.7298892736434937, "reward_total_composite_std": 0.14072683453559875} {"timestamp_utc": "2026-04-13T00:41:19Z", "mode": "train", "global_step": 979, "epoch": 0.09834254143646409, "loss": -0.045, "grad_norm": 15.671269416809082, "learning_rate": 7.036363636363637e-06, "num_tokens": 1764475.0, "completions/mean_length": 32.625, "completions/min_length": 24.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.625, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9929998517036438, "rewards/meter/std": 0.006410006899386644, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9290844798088074, "rewards/repeat_soft/std": 0.059195276349782944, "rewards/judge_quality/mean": 0.38374999165534973, "rewards/judge_quality/std": 0.11697832494974136, "rewards/total_composite/mean": 0.8048833608627319, "rewards/total_composite/std": 0.03488117456436157, "reward": 0.8048833608627319, "reward_std": 0.03488117456436157, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.112346351146698, "sampling/sampling_logp_difference/max": 1.1527141332626343, "sampling/importance_sampling_ratio/min": 0.31577855348587036, "sampling/importance_sampling_ratio/mean": 1.0186346769332886, "sampling/importance_sampling_ratio/max": 1.774185299873352, "entropy": 0.784535825252533, "clip_ratio/low_mean": 0.04539281874895096, "clip_ratio/low_min": 0.04539281874895096, "clip_ratio/high_mean": 0.06358762364834547, "clip_ratio/high_max": 0.06358762364834547, "clip_ratio/region_mean": 0.10898044239729643, "reward_total_mean": 0.8048833608627319, "reward_meter_mean": 0.9929998517036438, "reward_meter_std": 0.006410006899386644, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9290844798088074, "reward_repeat_soft_std": 0.059195276349782944, "reward_judge_quality_mean": 0.38374999165534973, "reward_judge_quality_std": 0.11697832494974136, "reward_total_composite_mean": 0.8048833608627319, "reward_total_composite_std": 0.03488117456436157} {"timestamp_utc": "2026-04-13T00:41:27Z", "mode": "train", "global_step": 980, "epoch": 0.09844299347061777, "loss": 0.1412, "grad_norm": 8.635858535766602, "learning_rate": 7.033333333333334e-06, "num_tokens": 1766384.0, "completions/mean_length": 59.625, "completions/min_length": 49.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.625, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9920778274536133, "rewards/meter/std": 0.0044211954809725285, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6444125771522522, "rewards/repeat_soft/std": 0.26895031332969666, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.12631450593471527, "rewards/total_composite/mean": 0.722751259803772, "rewards/total_composite/std": 0.05008711665868759, "reward": 0.722751259803772, "reward_std": 0.05008712410926819, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1011117473244667, "sampling/sampling_logp_difference/max": 1.8303375244140625, "sampling/importance_sampling_ratio/min": 0.1603594273328781, "sampling/importance_sampling_ratio/mean": 1.0121265649795532, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7680403627455235, "clip_ratio/low_mean": 0.03571412549354136, "clip_ratio/low_min": 0.03571412549354136, "clip_ratio/high_mean": 0.0630569476634264, "clip_ratio/high_max": 0.0630569476634264, "clip_ratio/region_mean": 0.09877107315696776, "reward_total_mean": 0.722751259803772, "reward_meter_mean": 0.9920778274536133, "reward_meter_std": 0.0044211954809725285, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6444125771522522, "reward_repeat_soft_std": 0.26895031332969666, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.12631450593471527, "reward_total_composite_mean": 0.722751259803772, "reward_total_composite_std": 0.05008711665868759} {"timestamp_utc": "2026-04-13T00:41:35Z", "mode": "train", "global_step": 981, "epoch": 0.09854344550477147, "loss": -0.0345, "grad_norm": 13.329865455627441, "learning_rate": 7.030303030303031e-06, "num_tokens": 1767958.0, "completions/mean_length": 47.75, "completions/min_length": 34.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.75, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.8093246817588806, "rewards/meter/std": 0.2811771035194397, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9885632991790771, "rewards/repeat_soft/std": 0.013217446394264698, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.7690525054931641, "rewards/total_composite/std": 0.1557956039905548, "reward": 0.7690525054931641, "reward_std": 0.15579558908939362, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16423466801643372, "sampling/sampling_logp_difference/max": 1.853492259979248, "sampling/importance_sampling_ratio/min": 0.15668901801109314, "sampling/importance_sampling_ratio/mean": 1.0263746976852417, "sampling/importance_sampling_ratio/max": 1.9016015529632568, "entropy": 1.346355862915516, "clip_ratio/low_mean": 0.047647325322031975, "clip_ratio/low_min": 0.047647325322031975, "clip_ratio/high_mean": 0.08404212445020676, "clip_ratio/high_max": 0.08404212445020676, "clip_ratio/region_mean": 0.13168944977223873, "reward_total_mean": 0.7690525054931641, "reward_meter_mean": 0.8093246817588806, "reward_meter_std": 0.2811771035194397, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9885632991790771, "reward_repeat_soft_std": 0.013217446394264698, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.7690525054931641, "reward_total_composite_std": 0.1557956039905548} {"timestamp_utc": "2026-04-13T00:41:48Z", "mode": "train", "global_step": 982, "epoch": 0.09864389753892516, "loss": 0.015, "grad_norm": 13.593945503234863, "learning_rate": 7.027272727272728e-06, "num_tokens": 1769644.0, "completions/mean_length": 48.75, "completions/min_length": 41.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.75, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9699594974517822, "rewards/meter/std": 0.03924940899014473, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9651694297790527, "rewards/repeat_soft/std": 0.03434298187494278, "rewards/judge_quality/mean": 0.5187499523162842, "rewards/judge_quality/std": 0.17908000946044922, "rewards/total_composite/mean": 0.8386236429214478, "rewards/total_composite/std": 0.0542929470539093, "reward": 0.8386236429214478, "reward_std": 0.054292943328619, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16092011332511902, "sampling/sampling_logp_difference/max": 1.3282403945922852, "sampling/importance_sampling_ratio/min": 0.26494303345680237, "sampling/importance_sampling_ratio/mean": 1.0034494400024414, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2152704671025276, "clip_ratio/low_mean": 0.11208624765276909, "clip_ratio/low_min": 0.11208624765276909, "clip_ratio/high_mean": 0.04074754845350981, "clip_ratio/high_max": 0.04074754845350981, "clip_ratio/region_mean": 0.1528337961062789, "reward_total_mean": 0.8386236429214478, "reward_meter_mean": 0.9699594974517822, "reward_meter_std": 0.03924940899014473, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9651694297790527, "reward_repeat_soft_std": 0.03434298187494278, "reward_judge_quality_mean": 0.5187499523162842, "reward_judge_quality_std": 0.17908000946044922, "reward_total_composite_mean": 0.8386236429214478, "reward_total_composite_std": 0.0542929470539093} {"timestamp_utc": "2026-04-13T00:41:57Z", "mode": "train", "global_step": 983, "epoch": 0.09874434957307886, "loss": 0.0915, "grad_norm": 23.710010528564453, "learning_rate": 7.024242424242424e-06, "num_tokens": 1771629.0, "completions/mean_length": 65.125, "completions/min_length": 55.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.125, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9356060028076172, "rewards/meter/std": 0.155043363571167, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7768068313598633, "rewards/repeat_soft/std": 0.10552231222391129, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.7495784163475037, "rewards/total_composite/std": 0.10666467994451523, "reward": 0.7495784163475037, "reward_std": 0.10666466504335403, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12439888715744019, "sampling/sampling_logp_difference/max": 1.6646509170532227, "sampling/importance_sampling_ratio/min": 0.1892567127943039, "sampling/importance_sampling_ratio/mean": 1.0056082010269165, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.841818057000637, "clip_ratio/low_mean": 0.028827750124037266, "clip_ratio/low_min": 0.028827750124037266, "clip_ratio/high_mean": 0.09460157807916403, "clip_ratio/high_max": 0.09460157807916403, "clip_ratio/region_mean": 0.1234293282032013, "reward_total_mean": 0.7495784163475037, "reward_meter_mean": 0.9356060028076172, "reward_meter_std": 0.155043363571167, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7768068313598633, "reward_repeat_soft_std": 0.10552231222391129, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.7495784163475037, "reward_total_composite_std": 0.10666467994451523} {"timestamp_utc": "2026-04-13T00:42:09Z", "mode": "train", "global_step": 984, "epoch": 0.09884480160723255, "loss": -0.0309, "grad_norm": 6.299431800842285, "learning_rate": 7.021212121212122e-06, "num_tokens": 1773039.0, "completions/mean_length": 88.25, "completions/min_length": 23.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 27.71428680419922, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.6942201852798462, "rewards/meter/std": 0.2585393190383911, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9610751867294312, "rewards/repeat_soft/std": 0.09121805429458618, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.2876474857330322, "rewards/total_composite/mean": 0.7103816270828247, "rewards/total_composite/std": 0.13931693136692047, "reward": 0.7103816270828247, "reward_std": 0.13931694626808167, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1936349719762802, "sampling/sampling_logp_difference/max": 2.0058889389038086, "sampling/importance_sampling_ratio/min": 0.1345406472682953, "sampling/importance_sampling_ratio/mean": 1.0234477519989014, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0728725269436836, "clip_ratio/low_mean": 0.0467391312122345, "clip_ratio/low_min": 0.0467391312122345, "clip_ratio/high_mean": 0.12425524927675724, "clip_ratio/high_max": 0.12425524927675724, "clip_ratio/region_mean": 0.17099438048899174, "reward_total_mean": 0.7103816270828247, "reward_meter_mean": 0.6942201852798462, "reward_meter_std": 0.2585393190383911, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9610751867294312, "reward_repeat_soft_std": 0.09121805429458618, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.2876474857330322, "reward_total_composite_mean": 0.7103816270828247, "reward_total_composite_std": 0.13931693136692047} {"timestamp_utc": "2026-04-13T00:42:17Z", "mode": "train", "global_step": 985, "epoch": 0.09894525364138623, "loss": -0.0255, "grad_norm": 14.421360969543457, "learning_rate": 7.018181818181818e-06, "num_tokens": 1774761.0, "completions/mean_length": 39.25, "completions/min_length": 31.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.25, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.7709047794342041, "rewards/meter/std": 0.29907968640327454, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.11785111576318741, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8877581357955933, "rewards/repeat_soft/std": 0.106243796646595, "rewards/judge_quality/mean": 0.27125000953674316, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.6233079433441162, "rewards/total_composite/std": 0.1262308955192566, "reward": 0.6233079433441162, "reward_std": 0.12623091042041779, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14539463818073273, "sampling/sampling_logp_difference/max": 2.1073760986328125, "sampling/importance_sampling_ratio/min": 0.12155649811029434, "sampling/importance_sampling_ratio/mean": 1.0296589136123657, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0774103328585625, "clip_ratio/low_mean": 0.05120091885328293, "clip_ratio/low_min": 0.05120091885328293, "clip_ratio/high_mean": 0.08239463064819574, "clip_ratio/high_max": 0.08239463064819574, "clip_ratio/region_mean": 0.13359554950147867, "reward_total_mean": 0.6233079433441162, "reward_meter_mean": 0.7709047794342041, "reward_meter_std": 0.29907968640327454, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.11785111576318741, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8877581357955933, "reward_repeat_soft_std": 0.106243796646595, "reward_judge_quality_mean": 0.27125000953674316, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.6233079433441162, "reward_total_composite_std": 0.1262308955192566} {"timestamp_utc": "2026-04-13T00:42:30Z", "mode": "train", "global_step": 986, "epoch": 0.09904570567553993, "loss": -0.1181, "grad_norm": 2.1593592166900635, "learning_rate": 7.015151515151516e-06, "num_tokens": 1776322.0, "completions/mean_length": 168.125, "completions/min_length": 50.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 53.5, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.45844748616218567, "rewards/meter/std": 0.34748029708862305, "rewards/count_adherence/mean": 0.59375, "rewards/count_adherence/std": 0.29693374037742615, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9878612160682678, "rewards/repeat_soft/std": 0.013733725063502789, "rewards/judge_quality/mean": 0.3062500059604645, "rewards/judge_quality/std": 0.1686871498823166, "rewards/total_composite/mean": 0.4526285231113434, "rewards/total_composite/std": 0.2944753170013428, "reward": 0.4526285231113434, "reward_std": 0.2944753170013428, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19160518050193787, "sampling/sampling_logp_difference/max": 1.4654865264892578, "sampling/importance_sampling_ratio/min": 0.23096559941768646, "sampling/importance_sampling_ratio/mean": 1.032717227935791, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4154329597949982, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.16137696988880634, "clip_ratio/high_max": 0.16137696988880634, "clip_ratio/region_mean": 0.16137696988880634, "reward_total_mean": 0.4526285231113434, "reward_meter_mean": 0.45844748616218567, "reward_meter_std": 0.34748029708862305, "reward_count_adherence_mean": 0.59375, "reward_count_adherence_std": 0.29693374037742615, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9878612160682678, "reward_repeat_soft_std": 0.013733725063502789, "reward_judge_quality_mean": 0.3062500059604645, "reward_judge_quality_std": 0.1686871498823166, "reward_total_composite_mean": 0.4526285231113434, "reward_total_composite_std": 0.2944753170013428} {"timestamp_utc": "2026-04-13T00:42:38Z", "mode": "train", "global_step": 987, "epoch": 0.09914615770969362, "loss": 0.0528, "grad_norm": 12.02418327331543, "learning_rate": 7.0121212121212126e-06, "num_tokens": 1778064.0, "completions/mean_length": 54.75, "completions/min_length": 47.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.75, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9700586795806885, "rewards/meter/std": 0.036086611449718475, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9074351787567139, "rewards/repeat_soft/std": 0.09577776491641998, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.22403763234615326, "rewards/total_composite/mean": 0.7680199146270752, "rewards/total_composite/std": 0.06719478219747543, "reward": 0.7680199146270752, "reward_std": 0.06719477474689484, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14233389496803284, "sampling/sampling_logp_difference/max": 1.2010531425476074, "sampling/importance_sampling_ratio/min": 0.3008771538734436, "sampling/importance_sampling_ratio/mean": 1.0390101671218872, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2544873505830765, "clip_ratio/low_mean": 0.06341656856238842, "clip_ratio/low_min": 0.06341656856238842, "clip_ratio/high_mean": 0.051989988423883915, "clip_ratio/high_max": 0.051989988423883915, "clip_ratio/region_mean": 0.11540655698627234, "reward_total_mean": 0.7680199146270752, "reward_meter_mean": 0.9700586795806885, "reward_meter_std": 0.036086611449718475, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9074351787567139, "reward_repeat_soft_std": 0.09577776491641998, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.22403763234615326, "reward_total_composite_mean": 0.7680199146270752, "reward_total_composite_std": 0.06719478219747543} {"timestamp_utc": "2026-04-13T00:42:46Z", "mode": "train", "global_step": 988, "epoch": 0.09924660974384732, "loss": 0.0405, "grad_norm": 10.927671432495117, "learning_rate": 7.00909090909091e-06, "num_tokens": 1779989.0, "completions/mean_length": 71.625, "completions/min_length": 49.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.625, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.9868839979171753, "rewards/meter/std": 0.014428826048970222, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8592594861984253, "rewards/repeat_soft/std": 0.03235481306910515, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7621487379074097, "rewards/total_composite/std": 0.018846627324819565, "reward": 0.7621487379074097, "reward_std": 0.01884663663804531, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13892808556556702, "sampling/sampling_logp_difference/max": 1.70357084274292, "sampling/importance_sampling_ratio/min": 0.18203234672546387, "sampling/importance_sampling_ratio/mean": 1.008994698524475, "sampling/importance_sampling_ratio/max": 1.9937056303024292, "entropy": 1.0191714987158775, "clip_ratio/low_mean": 0.02513979095965624, "clip_ratio/low_min": 0.02513979095965624, "clip_ratio/high_mean": 0.11504723038524389, "clip_ratio/high_max": 0.11504723038524389, "clip_ratio/region_mean": 0.14018702134490013, "reward_total_mean": 0.7621487379074097, "reward_meter_mean": 0.9868839979171753, "reward_meter_std": 0.014428826048970222, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8592594861984253, "reward_repeat_soft_std": 0.03235481306910515, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7621487379074097, "reward_total_composite_std": 0.018846627324819565} {"timestamp_utc": "2026-04-13T00:42:54Z", "mode": "train", "global_step": 989, "epoch": 0.09934706177800101, "loss": -0.0225, "grad_norm": 20.278236389160156, "learning_rate": 7.006060606060606e-06, "num_tokens": 1781413.0, "completions/mean_length": 29.0, "completions/min_length": 25.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.590722382068634, "rewards/meter/std": 0.3215080499649048, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9958781003952026, "rewards/repeat_soft/std": 0.004650620743632317, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.7164129018783569, "rewards/total_composite/std": 0.1646990329027176, "reward": 0.7164129018783569, "reward_std": 0.16469904780387878, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1624908447265625, "sampling/sampling_logp_difference/max": 1.773277759552002, "sampling/importance_sampling_ratio/min": 0.16977559030056, "sampling/importance_sampling_ratio/mean": 0.9752432703971863, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7438002936542034, "clip_ratio/low_mean": 0.04462780570611358, "clip_ratio/low_min": 0.04462780570611358, "clip_ratio/high_mean": 0.05429007392376661, "clip_ratio/high_max": 0.05429007392376661, "clip_ratio/region_mean": 0.09891787962988019, "reward_total_mean": 0.7164129018783569, "reward_meter_mean": 0.590722382068634, "reward_meter_std": 0.3215080499649048, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9958781003952026, "reward_repeat_soft_std": 0.004650620743632317, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.7164129018783569, "reward_total_composite_std": 0.1646990329027176} {"timestamp_utc": "2026-04-13T00:43:04Z", "mode": "train", "global_step": 990, "epoch": 0.09944751381215469, "loss": -0.0352, "grad_norm": 8.520520210266113, "learning_rate": 7.0030303030303035e-06, "num_tokens": 1784203.0, "completions/mean_length": 139.75, "completions/min_length": 119.0, "completions/max_length": 170.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 139.75, "completions/min_terminated_length": 119.0, "completions/max_terminated_length": 170.0, "rewards/meter/mean": 0.9796284437179565, "rewards/meter/std": 0.03327009454369545, "rewards/count_adherence/mean": 0.8541666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.5454009771347046, "rewards/repeat_soft/std": 0.16138580441474915, "rewards/judge_quality/mean": 0.14375001192092896, "rewards/judge_quality/std": 0.0176776722073555, "rewards/total_composite/mean": 0.6666228771209717, "rewards/total_composite/std": 0.014999181032180786, "reward": 0.6666228771209717, "reward_std": 0.014999188482761383, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09054137766361237, "sampling/sampling_logp_difference/max": 4.372806549072266, "sampling/importance_sampling_ratio/min": 0.01261578407138586, "sampling/importance_sampling_ratio/mean": 1.0194244384765625, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6537056043744087, "clip_ratio/low_mean": 0.024561191210523248, "clip_ratio/low_min": 0.024561191210523248, "clip_ratio/high_mean": 0.04786862898617983, "clip_ratio/high_max": 0.04786862898617983, "clip_ratio/region_mean": 0.07242982019670308, "reward_total_mean": 0.6666228771209717, "reward_meter_mean": 0.9796284437179565, "reward_meter_std": 0.03327009454369545, "reward_count_adherence_mean": 0.8541666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.5454009771347046, "reward_repeat_soft_std": 0.16138580441474915, "reward_judge_quality_mean": 0.14375001192092896, "reward_judge_quality_std": 0.0176776722073555, "reward_total_composite_mean": 0.6666228771209717, "reward_total_composite_std": 0.014999181032180786} {"timestamp_utc": "2026-04-13T00:43:17Z", "mode": "train", "global_step": 991, "epoch": 0.09954796584630839, "loss": -0.0784, "grad_norm": 2.2456445693969727, "learning_rate": 7e-06, "num_tokens": 1785500.0, "completions/mean_length": 84.125, "completions/min_length": 19.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 23.000001907348633, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.7723280191421509, "rewards/meter/std": 0.35973310470581055, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9478329420089722, "rewards/repeat_soft/std": 0.037053897976875305, "rewards/judge_quality/mean": 0.30125001072883606, "rewards/judge_quality/std": 0.14913439750671387, "rewards/total_composite/mean": 0.6500223875045776, "rewards/total_composite/std": 0.2782915234565735, "reward": 0.6500223875045776, "reward_std": 0.2782915234565735, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20179787278175354, "sampling/sampling_logp_difference/max": 1.4289112091064453, "sampling/importance_sampling_ratio/min": 0.23956961929798126, "sampling/importance_sampling_ratio/mean": 1.011790156364441, "sampling/importance_sampling_ratio/max": 1.7911747694015503, "entropy": 1.4929888844490051, "clip_ratio/low_mean": 0.015625, "clip_ratio/low_min": 0.015625, "clip_ratio/high_mean": 0.14767133630812168, "clip_ratio/high_max": 0.14767133630812168, "clip_ratio/region_mean": 0.16329633630812168, "reward_total_mean": 0.6500223875045776, "reward_meter_mean": 0.7723280191421509, "reward_meter_std": 0.35973310470581055, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9478329420089722, "reward_repeat_soft_std": 0.037053897976875305, "reward_judge_quality_mean": 0.30125001072883606, "reward_judge_quality_std": 0.14913439750671387, "reward_total_composite_mean": 0.6500223875045776, "reward_total_composite_std": 0.2782915234565735} {"timestamp_utc": "2026-04-13T00:43:24Z", "mode": "train", "global_step": 992, "epoch": 0.09964841788046208, "loss": 0.0635, "grad_norm": 11.105788230895996, "learning_rate": 6.996969696969698e-06, "num_tokens": 1787273.0, "completions/mean_length": 52.625, "completions/min_length": 45.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.625, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.856186032295227, "rewards/meter/std": 0.2878166735172272, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7119336724281311, "rewards/repeat_soft/std": 0.23458774387836456, "rewards/judge_quality/mean": 0.48124998807907104, "rewards/judge_quality/std": 0.2820049524307251, "rewards/total_composite/mean": 0.7133520841598511, "rewards/total_composite/std": 0.15914055705070496, "reward": 0.7133520841598511, "reward_std": 0.15914055705070496, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14688561856746674, "sampling/sampling_logp_difference/max": 1.637678623199463, "sampling/importance_sampling_ratio/min": 0.194430872797966, "sampling/importance_sampling_ratio/mean": 1.0252771377563477, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1316189765930176, "clip_ratio/low_mean": 0.04113026801496744, "clip_ratio/low_min": 0.04113026801496744, "clip_ratio/high_mean": 0.0808844636194408, "clip_ratio/high_max": 0.0808844636194408, "clip_ratio/region_mean": 0.12201473163440824, "reward_total_mean": 0.7133520841598511, "reward_meter_mean": 0.856186032295227, "reward_meter_std": 0.2878166735172272, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7119336724281311, "reward_repeat_soft_std": 0.23458774387836456, "reward_judge_quality_mean": 0.48124998807907104, "reward_judge_quality_std": 0.2820049524307251, "reward_total_composite_mean": 0.7133520841598511, "reward_total_composite_std": 0.15914055705070496} {"timestamp_utc": "2026-04-13T00:43:36Z", "mode": "train", "global_step": 993, "epoch": 0.09974886991461578, "loss": -0.0468, "grad_norm": 2.977363348007202, "learning_rate": 6.993939393939394e-06, "num_tokens": 1788643.0, "completions/mean_length": 82.25, "completions/min_length": 15.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 20.85714340209961, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.3772317171096802, "rewards/meter/std": 0.3792693316936493, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9327490329742432, "rewards/repeat_soft/std": 0.053178273141384125, "rewards/judge_quality/mean": 0.3512499928474426, "rewards/judge_quality/std": 0.15797263383865356, "rewards/total_composite/mean": 0.48570895195007324, "rewards/total_composite/std": 0.2612736225128174, "reward": 0.48570895195007324, "reward_std": 0.2612736225128174, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19766367971897125, "sampling/sampling_logp_difference/max": 1.4372756481170654, "sampling/importance_sampling_ratio/min": 0.23757411539554596, "sampling/importance_sampling_ratio/mean": 1.045865774154663, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2772387564182281, "clip_ratio/low_mean": 0.03803053870797157, "clip_ratio/low_min": 0.03803053870797157, "clip_ratio/high_mean": 0.07525252643972635, "clip_ratio/high_max": 0.07525252643972635, "clip_ratio/region_mean": 0.11328306514769793, "reward_total_mean": 0.48570895195007324, "reward_meter_mean": 0.3772317171096802, "reward_meter_std": 0.3792693316936493, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9327490329742432, "reward_repeat_soft_std": 0.053178273141384125, "reward_judge_quality_mean": 0.3512499928474426, "reward_judge_quality_std": 0.15797263383865356, "reward_total_composite_mean": 0.48570895195007324, "reward_total_composite_std": 0.2612736225128174} {"timestamp_utc": "2026-04-13T00:43:49Z", "mode": "train", "global_step": 994, "epoch": 0.09984932194876946, "loss": -0.0896, "grad_norm": 1.5982019901275635, "learning_rate": 6.990909090909092e-06, "num_tokens": 1790163.0, "completions/mean_length": 220.0, "completions/min_length": 42.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 44.79999923706055, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.7331694960594177, "rewards/meter/std": 0.3724997639656067, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.1178511381149292, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9794262647628784, "rewards/repeat_soft/std": 0.02454124204814434, "rewards/judge_quality/mean": 0.32625001668930054, "rewards/judge_quality/std": 0.29765453934669495, "rewards/total_composite/mean": 0.4927855134010315, "rewards/total_composite/std": 0.4124957323074341, "reward": 0.4927855134010315, "reward_std": 0.41249576210975647, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16564172506332397, "sampling/sampling_logp_difference/max": 1.356736183166504, "sampling/importance_sampling_ratio/min": 0.2574998438358307, "sampling/importance_sampling_ratio/mean": 1.0442867279052734, "sampling/importance_sampling_ratio/max": 1.9955836534500122, "entropy": 0.9104850515723228, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.1072813868522644, "clip_ratio/high_max": 0.1072813868522644, "clip_ratio/region_mean": 0.1072813868522644, "reward_total_mean": 0.4927855134010315, "reward_meter_mean": 0.7331694960594177, "reward_meter_std": 0.3724997639656067, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.1178511381149292, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9794262647628784, "reward_repeat_soft_std": 0.02454124204814434, "reward_judge_quality_mean": 0.32625001668930054, "reward_judge_quality_std": 0.29765453934669495, "reward_total_composite_mean": 0.4927855134010315, "reward_total_composite_std": 0.4124957323074341} {"timestamp_utc": "2026-04-13T00:43:56Z", "mode": "train", "global_step": 995, "epoch": 0.09994977398292315, "loss": 0.0268, "grad_norm": 12.625038146972656, "learning_rate": 6.987878787878788e-06, "num_tokens": 1791681.0, "completions/mean_length": 25.75, "completions/min_length": 22.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.75, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.6767120957374573, "rewards/meter/std": 0.28353509306907654, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9828473329544067, "rewards/repeat_soft/std": 0.03345540165901184, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.6986801624298096, "rewards/total_composite/std": 0.1153237447142601, "reward": 0.6986801624298096, "reward_std": 0.1153237447142601, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13255822658538818, "sampling/sampling_logp_difference/max": 1.6344332695007324, "sampling/importance_sampling_ratio/min": 0.19506287574768066, "sampling/importance_sampling_ratio/mean": 0.9954116940498352, "sampling/importance_sampling_ratio/max": 1.7202554941177368, "entropy": 0.8152746260166168, "clip_ratio/low_mean": 0.054624596145004034, "clip_ratio/low_min": 0.054624596145004034, "clip_ratio/high_mean": 0.11611070390790701, "clip_ratio/high_max": 0.11611070390790701, "clip_ratio/region_mean": 0.17073530005291104, "reward_total_mean": 0.6986801624298096, "reward_meter_mean": 0.6767120957374573, "reward_meter_std": 0.28353509306907654, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9828473329544067, "reward_repeat_soft_std": 0.03345540165901184, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.6986801624298096, "reward_total_composite_std": 0.1153237447142601} {"timestamp_utc": "2026-04-13T00:44:02Z", "mode": "train", "global_step": 996, "epoch": 0.10005022601707685, "loss": 0.0884, "grad_norm": 25.639385223388672, "learning_rate": 6.984848484848485e-06, "num_tokens": 1793163.0, "completions/mean_length": 22.25, "completions/min_length": 18.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.25, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.8462362289428711, "rewards/meter/std": 0.2289504110813141, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.6262500286102295, "rewards/judge_quality/std": 0.2432481348514557, "rewards/total_composite/mean": 0.8149312734603882, "rewards/total_composite/std": 0.1381424218416214, "reward": 0.8149312734603882, "reward_std": 0.1381424218416214, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1937328279018402, "sampling/sampling_logp_difference/max": 1.594740867614746, "sampling/importance_sampling_ratio/min": 0.20296113193035126, "sampling/importance_sampling_ratio/mean": 0.994759738445282, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.307232953608036, "clip_ratio/low_mean": 0.05240384861826897, "clip_ratio/low_min": 0.05240384861826897, "clip_ratio/high_mean": 0.13618718646466732, "clip_ratio/high_max": 0.13618718646466732, "clip_ratio/region_mean": 0.1885910350829363, "reward_total_mean": 0.8149312734603882, "reward_meter_mean": 0.8462362289428711, "reward_meter_std": 0.2289504110813141, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.6262500286102295, "reward_judge_quality_std": 0.2432481348514557, "reward_total_composite_mean": 0.8149312734603882, "reward_total_composite_std": 0.1381424218416214} {"timestamp_utc": "2026-04-13T00:44:10Z", "mode": "train", "global_step": 997, "epoch": 0.10015067805123054, "loss": -0.0098, "grad_norm": 12.113319396972656, "learning_rate": 6.981818181818183e-06, "num_tokens": 1794660.0, "completions/mean_length": 37.125, "completions/min_length": 31.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.125, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.9936320781707764, "rewards/meter/std": 0.0008012447506189346, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8458464741706848, "rewards/repeat_soft/std": 0.1648925095796585, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.8035941123962402, "rewards/total_composite/std": 0.0313386544585228, "reward": 0.8035941123962402, "reward_std": 0.031338661909103394, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11195699870586395, "sampling/sampling_logp_difference/max": 0.8812623023986816, "sampling/importance_sampling_ratio/min": 0.414259672164917, "sampling/importance_sampling_ratio/mean": 1.0371545553207397, "sampling/importance_sampling_ratio/max": 1.9020394086837769, "entropy": 0.8509281724691391, "clip_ratio/low_mean": 0.02223190851509571, "clip_ratio/low_min": 0.02223190851509571, "clip_ratio/high_mean": 0.10743585880845785, "clip_ratio/high_max": 0.10743585880845785, "clip_ratio/region_mean": 0.12966776732355356, "reward_total_mean": 0.8035941123962402, "reward_meter_mean": 0.9936320781707764, "reward_meter_std": 0.0008012447506189346, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8458464741706848, "reward_repeat_soft_std": 0.1648925095796585, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.8035941123962402, "reward_total_composite_std": 0.0313386544585228} {"timestamp_utc": "2026-04-13T00:44:19Z", "mode": "train", "global_step": 998, "epoch": 0.10025113008538424, "loss": -0.0471, "grad_norm": 6.764216899871826, "learning_rate": 6.978787878787879e-06, "num_tokens": 1796741.0, "completions/mean_length": 90.125, "completions/min_length": 61.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.125, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.9850067496299744, "rewards/meter/std": 0.016317695379257202, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6476006507873535, "rewards/repeat_soft/std": 0.15330816805362701, "rewards/judge_quality/mean": 0.26375001668930054, "rewards/judge_quality/std": 0.13373079895973206, "rewards/total_composite/mean": 0.6996380686759949, "rewards/total_composite/std": 0.05142730101943016, "reward": 0.6996380686759949, "reward_std": 0.05142730474472046, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08262017369270325, "sampling/sampling_logp_difference/max": 1.6100382804870605, "sampling/importance_sampling_ratio/min": 0.1998799592256546, "sampling/importance_sampling_ratio/mean": 1.0086666345596313, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.657436303794384, "clip_ratio/low_mean": 0.039646593388170004, "clip_ratio/low_min": 0.039646593388170004, "clip_ratio/high_mean": 0.05000151041895151, "clip_ratio/high_max": 0.05000151041895151, "clip_ratio/region_mean": 0.08964810380712152, "reward_total_mean": 0.6996380686759949, "reward_meter_mean": 0.9850067496299744, "reward_meter_std": 0.016317695379257202, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6476006507873535, "reward_repeat_soft_std": 0.15330816805362701, "reward_judge_quality_mean": 0.26375001668930054, "reward_judge_quality_std": 0.13373079895973206, "reward_total_composite_mean": 0.6996380686759949, "reward_total_composite_std": 0.05142730101943016} {"timestamp_utc": "2026-04-13T00:44:27Z", "mode": "train", "global_step": 999, "epoch": 0.10035158211953792, "loss": 0.0172, "grad_norm": 11.75110149383545, "learning_rate": 6.975757575757577e-06, "num_tokens": 1798560.0, "completions/mean_length": 53.375, "completions/min_length": 46.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.375, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.989327073097229, "rewards/meter/std": 0.016622032970190048, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9021143913269043, "rewards/repeat_soft/std": 0.08183584362268448, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8114086389541626, "rewards/total_composite/std": 0.008895275183022022, "reward": 0.8114086389541626, "reward_std": 0.008895276114344597, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1429656744003296, "sampling/sampling_logp_difference/max": 1.8762903213500977, "sampling/importance_sampling_ratio/min": 0.15315721929073334, "sampling/importance_sampling_ratio/mean": 1.0202945470809937, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9619656801223755, "clip_ratio/low_mean": 0.034119898453354836, "clip_ratio/low_min": 0.034119898453354836, "clip_ratio/high_mean": 0.10237073805183172, "clip_ratio/high_max": 0.10237073805183172, "clip_ratio/region_mean": 0.13649063650518656, "reward_total_mean": 0.8114086389541626, "reward_meter_mean": 0.989327073097229, "reward_meter_std": 0.016622032970190048, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9021143913269043, "reward_repeat_soft_std": 0.08183584362268448, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8114086389541626, "reward_total_composite_std": 0.008895275183022022} {"timestamp_utc": "2026-04-13T00:44:37Z", "mode": "train", "global_step": 1000, "epoch": 0.10045203415369161, "loss": 0.0862, "grad_norm": 18.73358154296875, "learning_rate": 6.9727272727272735e-06, "num_tokens": 1800063.0, "completions/mean_length": 39.875, "completions/min_length": 33.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.7313731908798218, "rewards/meter/std": 0.37778565287590027, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9793304204940796, "rewards/repeat_soft/std": 0.03336813673377037, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.263435423374176, "rewards/total_composite/mean": 0.7291760444641113, "rewards/total_composite/std": 0.20406803488731384, "reward": 0.7291760444641113, "reward_std": 0.20406801998615265, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16711480915546417, "sampling/sampling_logp_difference/max": 2.4503016471862793, "sampling/importance_sampling_ratio/min": 0.08626756072044373, "sampling/importance_sampling_ratio/mean": 1.0327154397964478, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2966601699590683, "clip_ratio/low_mean": 0.041972565464675426, "clip_ratio/low_min": 0.041972565464675426, "clip_ratio/high_mean": 0.11219336465001106, "clip_ratio/high_max": 0.11219336465001106, "clip_ratio/region_mean": 0.1541659301146865, "reward_total_mean": 0.7291760444641113, "reward_meter_mean": 0.7313731908798218, "reward_meter_std": 0.37778565287590027, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9793304204940796, "reward_repeat_soft_std": 0.03336813673377037, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.263435423374176, "reward_total_composite_mean": 0.7291760444641113, "reward_total_composite_std": 0.20406803488731384} {"timestamp_utc": "2026-04-13T00:45:20Z", "mode": "eval", "global_step": 1000, "epoch": 0.10045203415369161, "eval_loss": NaN, "eval_runtime": 43.576, "eval_samples_per_second": 1.836, "eval_steps_per_second": 0.229, "eval_num_tokens": 1800063.0, "eval_completions/mean_length": 52.675, "eval_completions/min_length": 28.3, "eval_completions/max_length": 89.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 52.675, "eval_completions/min_terminated_length": 28.3, "eval_completions/max_terminated_length": 89.0, "eval_rewards/meter/mean": 0.8715584993362426, "eval_rewards/meter/std": 0.22732697036117316, "eval_rewards/count_adherence/mean": 0.8437500059604645, "eval_rewards/count_adherence/std": 0.11981934532523156, "eval_rewards/hard_gate/mean": 1.0, "eval_rewards/hard_gate/std": 0.0, "eval_rewards/repeat_soft/mean": 0.8621283650398255, "eval_rewards/repeat_soft/std": 0.12735465578734875, "eval_rewards/judge_quality/mean": 0.3793749988079071, "eval_rewards/judge_quality/std": 0.11860981611534953, "eval_rewards/total_composite/mean": 0.7187891662120819, "eval_rewards/total_composite/std": 0.11477440260350705, "eval_reward": 0.7187891662120819, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.07079344913363457, "eval_sampling/sampling_logp_difference/max": 1.0467300415039062, "eval_sampling/importance_sampling_ratio/min": 0.35709895342588427, "eval_sampling/importance_sampling_ratio/mean": 1.0189024806022644, "eval_sampling/importance_sampling_ratio/max": 1.4543935060501099, "eval_entropy": 0.8374313056468964, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7187891662120819, "eval_reward_meter_mean": 0.8715584993362426, "eval_reward_meter_std": 0.22732697036117316, "eval_reward_count_adherence_mean": 0.8437500059604645, "eval_reward_count_adherence_std": 0.11981934532523156, "eval_reward_hard_gate_mean": 1.0, "eval_reward_hard_gate_std": 0.0, "eval_reward_repeat_soft_mean": 0.8621283650398255, "eval_reward_repeat_soft_std": 0.12735465578734875, "eval_reward_judge_quality_mean": 0.3793749988079071, "eval_reward_judge_quality_std": 0.11860981611534953, "eval_reward_total_composite_mean": 0.7187891662120819, "eval_reward_total_composite_std": 0.11477440260350705} {"timestamp_utc": "2026-04-13T00:45:31Z", "mode": "train", "global_step": 1001, "epoch": 0.1005524861878453, "loss": 0.0579, "grad_norm": 7.8100080490112305, "learning_rate": 6.969696969696971e-06, "num_tokens": 1802272.0, "completions/mean_length": 82.125, "completions/min_length": 62.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.125, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.9886900186538696, "rewards/meter/std": 0.009890801273286343, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8364235162734985, "rewards/repeat_soft/std": 0.0679231509566307, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7618027925491333, "rewards/total_composite/std": 0.027247682213783264, "reward": 0.7618027925491333, "reward_std": 0.02724769338965416, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13539919257164001, "sampling/sampling_logp_difference/max": 2.7429094314575195, "sampling/importance_sampling_ratio/min": 0.06438275426626205, "sampling/importance_sampling_ratio/mean": 1.000666618347168, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7358815521001816, "clip_ratio/low_mean": 0.014926861505955458, "clip_ratio/low_min": 0.014926861505955458, "clip_ratio/high_mean": 0.11887802183628082, "clip_ratio/high_max": 0.11887802183628082, "clip_ratio/region_mean": 0.13380488334223628, "reward_total_mean": 0.7618027925491333, "reward_meter_mean": 0.9886900186538696, "reward_meter_std": 0.009890801273286343, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8364235162734985, "reward_repeat_soft_std": 0.0679231509566307, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7618027925491333, "reward_total_composite_std": 0.027247682213783264} {"timestamp_utc": "2026-04-13T00:45:38Z", "mode": "train", "global_step": 1002, "epoch": 0.100652938221999, "loss": 0.0706, "grad_norm": 15.538141250610352, "learning_rate": 6.966666666666667e-06, "num_tokens": 1803754.0, "completions/mean_length": 33.25, "completions/min_length": 28.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.25, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.8989349603652954, "rewards/meter/std": 0.24130313098430634, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8464912176132202, "rewards/repeat_soft/std": 0.08681753277778625, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7151699066162109, "rewards/total_composite/std": 0.1150195449590683, "reward": 0.7151699066162109, "reward_std": 0.11501955986022949, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14065054059028625, "sampling/sampling_logp_difference/max": 1.2598717212677002, "sampling/importance_sampling_ratio/min": 0.2836904227733612, "sampling/importance_sampling_ratio/mean": 1.0394514799118042, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9276437684893608, "clip_ratio/low_mean": 0.020270269364118576, "clip_ratio/low_min": 0.020270269364118576, "clip_ratio/high_mean": 0.12603861279785633, "clip_ratio/high_max": 0.12603861279785633, "clip_ratio/region_mean": 0.1463088821619749, "reward_total_mean": 0.7151699066162109, "reward_meter_mean": 0.8989349603652954, "reward_meter_std": 0.24130313098430634, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8464912176132202, "reward_repeat_soft_std": 0.08681753277778625, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7151699066162109, "reward_total_composite_std": 0.1150195449590683} {"timestamp_utc": "2026-04-13T00:45:45Z", "mode": "train", "global_step": 1003, "epoch": 0.10075339025615268, "loss": -0.0147, "grad_norm": 14.171834945678711, "learning_rate": 6.963636363636364e-06, "num_tokens": 1805485.0, "completions/mean_length": 37.375, "completions/min_length": 28.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.375, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.5510011911392212, "rewards/meter/std": 0.47137749195098877, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9132547974586487, "rewards/repeat_soft/std": 0.07379785925149918, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.6152759790420532, "rewards/total_composite/std": 0.21007221937179565, "reward": 0.6152759790420532, "reward_std": 0.21007221937179565, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16192246973514557, "sampling/sampling_logp_difference/max": 1.5309324264526367, "sampling/importance_sampling_ratio/min": 0.21633386611938477, "sampling/importance_sampling_ratio/mean": 1.018682837486267, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3240332454442978, "clip_ratio/low_mean": 0.0638184854760766, "clip_ratio/low_min": 0.0638184854760766, "clip_ratio/high_mean": 0.07152025494724512, "clip_ratio/high_max": 0.07152025494724512, "clip_ratio/region_mean": 0.13533874042332172, "reward_total_mean": 0.6152759790420532, "reward_meter_mean": 0.5510011911392212, "reward_meter_std": 0.47137749195098877, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9132547974586487, "reward_repeat_soft_std": 0.07379785925149918, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.6152759790420532, "reward_total_composite_std": 0.21007221937179565} {"timestamp_utc": "2026-04-13T00:45:53Z", "mode": "train", "global_step": 1004, "epoch": 0.10085384229030638, "loss": 0.0224, "grad_norm": 13.156856536865234, "learning_rate": 6.960606060606061e-06, "num_tokens": 1807193.0, "completions/mean_length": 54.5, "completions/min_length": 36.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.8060961365699768, "rewards/meter/std": 0.29258453845977783, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.718640923500061, "rewards/repeat_soft/std": 0.23446260392665863, "rewards/judge_quality/mean": 0.48124998807907104, "rewards/judge_quality/std": 0.22949868440628052, "rewards/total_composite/mean": 0.6914823651313782, "rewards/total_composite/std": 0.15366427600383759, "reward": 0.6914823651313782, "reward_std": 0.15366427600383759, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12539628148078918, "sampling/sampling_logp_difference/max": 1.517165184020996, "sampling/importance_sampling_ratio/min": 0.21933278441429138, "sampling/importance_sampling_ratio/mean": 1.0230882167816162, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8336143456399441, "clip_ratio/low_mean": 0.06564824096858501, "clip_ratio/low_min": 0.06564824096858501, "clip_ratio/high_mean": 0.06258220644667745, "clip_ratio/high_max": 0.06258220644667745, "clip_ratio/region_mean": 0.12823044741526246, "reward_total_mean": 0.6914823651313782, "reward_meter_mean": 0.8060961365699768, "reward_meter_std": 0.29258453845977783, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.718640923500061, "reward_repeat_soft_std": 0.23446260392665863, "reward_judge_quality_mean": 0.48124998807907104, "reward_judge_quality_std": 0.22949868440628052, "reward_total_composite_mean": 0.6914823651313782, "reward_total_composite_std": 0.15366427600383759} {"timestamp_utc": "2026-04-13T00:46:00Z", "mode": "train", "global_step": 1005, "epoch": 0.10095429432446007, "loss": 0.0414, "grad_norm": 17.659025192260742, "learning_rate": 6.957575757575759e-06, "num_tokens": 1808664.0, "completions/mean_length": 28.875, "completions/min_length": 25.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.875, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9661886692047119, "rewards/meter/std": 0.03310197591781616, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9388188123703003, "rewards/repeat_soft/std": 0.035052020102739334, "rewards/judge_quality/mean": 0.38374999165534973, "rewards/judge_quality/std": 0.11697831749916077, "rewards/total_composite/mean": 0.7937917709350586, "rewards/total_composite/std": 0.03259976580739021, "reward": 0.7937917709350586, "reward_std": 0.032599762082099915, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14028657972812653, "sampling/sampling_logp_difference/max": 1.25799560546875, "sampling/importance_sampling_ratio/min": 0.28422313928604126, "sampling/importance_sampling_ratio/mean": 0.9988425374031067, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7972437366843224, "clip_ratio/low_mean": 0.04612069111317396, "clip_ratio/low_min": 0.04612069111317396, "clip_ratio/high_mean": 0.10811392217874527, "clip_ratio/high_max": 0.10811392217874527, "clip_ratio/region_mean": 0.15423461329191923, "reward_total_mean": 0.7937917709350586, "reward_meter_mean": 0.9661886692047119, "reward_meter_std": 0.03310197591781616, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9388188123703003, "reward_repeat_soft_std": 0.035052020102739334, "reward_judge_quality_mean": 0.38374999165534973, "reward_judge_quality_std": 0.11697831749916077, "reward_total_composite_mean": 0.7937917709350586, "reward_total_composite_std": 0.03259976580739021} {"timestamp_utc": "2026-04-13T00:46:08Z", "mode": "train", "global_step": 1006, "epoch": 0.10105474635861376, "loss": -0.0226, "grad_norm": 13.342090606689453, "learning_rate": 6.954545454545455e-06, "num_tokens": 1810465.0, "completions/mean_length": 53.125, "completions/min_length": 44.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.125, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9835500121116638, "rewards/meter/std": 0.011465338058769703, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.966254472732544, "rewards/repeat_soft/std": 0.03404707834124565, "rewards/judge_quality/mean": 0.41874998807907104, "rewards/judge_quality/std": 0.14574319124221802, "rewards/total_composite/mean": 0.7648479342460632, "rewards/total_composite/std": 0.04501848295331001, "reward": 0.7648479342460632, "reward_std": 0.045018501579761505, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13322532176971436, "sampling/sampling_logp_difference/max": 1.2820427417755127, "sampling/importance_sampling_ratio/min": 0.2774699330329895, "sampling/importance_sampling_ratio/mean": 1.0179708003997803, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8511311933398247, "clip_ratio/low_mean": 0.07907003909349442, "clip_ratio/low_min": 0.07907003909349442, "clip_ratio/high_mean": 0.04202179331332445, "clip_ratio/high_max": 0.04202179331332445, "clip_ratio/region_mean": 0.12109183240681887, "reward_total_mean": 0.7648479342460632, "reward_meter_mean": 0.9835500121116638, "reward_meter_std": 0.011465338058769703, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.966254472732544, "reward_repeat_soft_std": 0.03404707834124565, "reward_judge_quality_mean": 0.41874998807907104, "reward_judge_quality_std": 0.14574319124221802, "reward_total_composite_mean": 0.7648479342460632, "reward_total_composite_std": 0.04501848295331001} {"timestamp_utc": "2026-04-13T00:46:15Z", "mode": "train", "global_step": 1007, "epoch": 0.10115519839276746, "loss": 0.0322, "grad_norm": 17.444425582885742, "learning_rate": 6.951515151515153e-06, "num_tokens": 1811957.0, "completions/mean_length": 32.5, "completions/min_length": 30.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.5, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.7230027318000793, "rewards/meter/std": 0.41347917914390564, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9553639888763428, "rewards/repeat_soft/std": 0.03395828232169151, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.7178876399993896, "rewards/total_composite/std": 0.20758786797523499, "reward": 0.7178876399993896, "reward_std": 0.2075878530740738, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15721793472766876, "sampling/sampling_logp_difference/max": 1.6123051643371582, "sampling/importance_sampling_ratio/min": 0.19942736625671387, "sampling/importance_sampling_ratio/mean": 1.0143091678619385, "sampling/importance_sampling_ratio/max": 1.887968897819519, "entropy": 1.1816636845469475, "clip_ratio/low_mean": 0.03515625, "clip_ratio/low_min": 0.03515625, "clip_ratio/high_mean": 0.1316699581220746, "clip_ratio/high_max": 0.1316699581220746, "clip_ratio/region_mean": 0.1668262081220746, "reward_total_mean": 0.7178876399993896, "reward_meter_mean": 0.7230027318000793, "reward_meter_std": 0.41347917914390564, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9553639888763428, "reward_repeat_soft_std": 0.03395828232169151, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.7178876399993896, "reward_total_composite_std": 0.20758786797523499} {"timestamp_utc": "2026-04-13T00:46:24Z", "mode": "train", "global_step": 1008, "epoch": 0.10125565042692114, "loss": 0.0461, "grad_norm": 13.93464469909668, "learning_rate": 6.948484848484849e-06, "num_tokens": 1813775.0, "completions/mean_length": 51.25, "completions/min_length": 42.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.25, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9053633213043213, "rewards/meter/std": 0.16360773146152496, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8824666738510132, "rewards/repeat_soft/std": 0.09877054393291473, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7727851867675781, "rewards/total_composite/std": 0.07263234257698059, "reward": 0.7727851867675781, "reward_std": 0.07263235002756119, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15889066457748413, "sampling/sampling_logp_difference/max": 2.336836814880371, "sampling/importance_sampling_ratio/min": 0.09663282334804535, "sampling/importance_sampling_ratio/mean": 0.9989093542098999, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7679200917482376, "clip_ratio/low_mean": 0.03299825266003609, "clip_ratio/low_min": 0.03299825266003609, "clip_ratio/high_mean": 0.10314019909128547, "clip_ratio/high_max": 0.10314019909128547, "clip_ratio/region_mean": 0.13613845175132155, "reward_total_mean": 0.7727851867675781, "reward_meter_mean": 0.9053633213043213, "reward_meter_std": 0.16360773146152496, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8824666738510132, "reward_repeat_soft_std": 0.09877054393291473, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7727851867675781, "reward_total_composite_std": 0.07263234257698059} {"timestamp_utc": "2026-04-13T00:46:31Z", "mode": "train", "global_step": 1009, "epoch": 0.10135610246107483, "loss": 0.0318, "grad_norm": 19.815265655517578, "learning_rate": 6.945454545454546e-06, "num_tokens": 1815472.0, "completions/mean_length": 46.125, "completions/min_length": 37.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.125, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.9908621311187744, "rewards/meter/std": 0.004093292634934187, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.851320743560791, "rewards/repeat_soft/std": 0.15331174433231354, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.8006450533866882, "rewards/total_composite/std": 0.025484347715973854, "reward": 0.8006450533866882, "reward_std": 0.025484351441264153, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1299409717321396, "sampling/sampling_logp_difference/max": 1.2520718574523926, "sampling/importance_sampling_ratio/min": 0.28591182827949524, "sampling/importance_sampling_ratio/mean": 1.018559455871582, "sampling/importance_sampling_ratio/max": 1.9075405597686768, "entropy": 0.9232023134827614, "clip_ratio/low_mean": 0.015964673832058907, "clip_ratio/low_min": 0.015964673832058907, "clip_ratio/high_mean": 0.11546137183904648, "clip_ratio/high_max": 0.11546137183904648, "clip_ratio/region_mean": 0.13142604567110538, "reward_total_mean": 0.8006450533866882, "reward_meter_mean": 0.9908621311187744, "reward_meter_std": 0.004093292634934187, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.851320743560791, "reward_repeat_soft_std": 0.15331174433231354, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.8006450533866882, "reward_total_composite_std": 0.025484347715973854} {"timestamp_utc": "2026-04-13T00:46:40Z", "mode": "train", "global_step": 1010, "epoch": 0.10145655449522853, "loss": -0.0982, "grad_norm": 5.6405720710754395, "learning_rate": 6.942424242424243e-06, "num_tokens": 1818050.0, "completions/mean_length": 110.25, "completions/min_length": 64.0, "completions/max_length": 139.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.25, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.9950304627418518, "rewards/meter/std": 0.0028495800215750933, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.07715165615081787, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.5741122364997864, "rewards/repeat_soft/std": 0.20561227202415466, "rewards/judge_quality/mean": 0.23000000417232513, "rewards/judge_quality/std": 0.12224097549915314, "rewards/total_composite/mean": 0.6804249286651611, "rewards/total_composite/std": 0.05506591498851776, "reward": 0.6804249286651611, "reward_std": 0.05506591126322746, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07866483926773071, "sampling/sampling_logp_difference/max": 1.1464643478393555, "sampling/importance_sampling_ratio/min": 0.3177582621574402, "sampling/importance_sampling_ratio/mean": 1.0119166374206543, "sampling/importance_sampling_ratio/max": 1.9199362993240356, "entropy": 0.5451361611485481, "clip_ratio/low_mean": 0.045266075525432825, "clip_ratio/low_min": 0.045266075525432825, "clip_ratio/high_mean": 0.035969334188848734, "clip_ratio/high_max": 0.035969334188848734, "clip_ratio/region_mean": 0.08123540971428156, "reward_total_mean": 0.6804249286651611, "reward_meter_mean": 0.9950304627418518, "reward_meter_std": 0.0028495800215750933, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.07715165615081787, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.5741122364997864, "reward_repeat_soft_std": 0.20561227202415466, "reward_judge_quality_mean": 0.23000000417232513, "reward_judge_quality_std": 0.12224097549915314, "reward_total_composite_mean": 0.6804249286651611, "reward_total_composite_std": 0.05506591498851776} {"timestamp_utc": "2026-04-13T00:46:48Z", "mode": "train", "global_step": 1011, "epoch": 0.10155700652938222, "loss": -0.0047, "grad_norm": 8.38034725189209, "learning_rate": 6.93939393939394e-06, "num_tokens": 1820206.0, "completions/mean_length": 73.5, "completions/min_length": 51.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.5, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.9921414256095886, "rewards/meter/std": 0.0055219619534909725, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7243751883506775, "rewards/repeat_soft/std": 0.18488678336143494, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.7319011688232422, "rewards/total_composite/std": 0.03189864009618759, "reward": 0.7319011688232422, "reward_std": 0.03189864382147789, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12059754878282547, "sampling/sampling_logp_difference/max": 5.320353031158447, "sampling/importance_sampling_ratio/min": 0.0048910267651081085, "sampling/importance_sampling_ratio/mean": 1.0003875494003296, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6881185211241245, "clip_ratio/low_mean": 0.04309694771654904, "clip_ratio/low_min": 0.04309694771654904, "clip_ratio/high_mean": 0.06311335694044828, "clip_ratio/high_max": 0.06311335694044828, "clip_ratio/region_mean": 0.10621030465699732, "reward_total_mean": 0.7319011688232422, "reward_meter_mean": 0.9921414256095886, "reward_meter_std": 0.0055219619534909725, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7243751883506775, "reward_repeat_soft_std": 0.18488678336143494, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.7319011688232422, "reward_total_composite_std": 0.03189864009618759} {"timestamp_utc": "2026-04-13T00:46:54Z", "mode": "train", "global_step": 1012, "epoch": 0.10165745856353592, "loss": 0.0339, "grad_norm": 14.752399444580078, "learning_rate": 6.936363636363636e-06, "num_tokens": 1821687.0, "completions/mean_length": 34.125, "completions/min_length": 32.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.125, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.5698710680007935, "rewards/meter/std": 0.45105212926864624, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9224269986152649, "rewards/repeat_soft/std": 0.05293147638440132, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.6205596923828125, "rewards/total_composite/std": 0.19636744260787964, "reward": 0.6205596923828125, "reward_std": 0.19636744260787964, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13898847997188568, "sampling/sampling_logp_difference/max": 1.7038874626159668, "sampling/importance_sampling_ratio/min": 0.18197472393512726, "sampling/importance_sampling_ratio/mean": 1.0167959928512573, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0274399891495705, "clip_ratio/low_mean": 0.057631161995232105, "clip_ratio/low_min": 0.057631161995232105, "clip_ratio/high_mean": 0.06360479909926653, "clip_ratio/high_max": 0.06360479909926653, "clip_ratio/region_mean": 0.12123596109449863, "reward_total_mean": 0.6205596923828125, "reward_meter_mean": 0.5698710680007935, "reward_meter_std": 0.45105212926864624, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9224269986152649, "reward_repeat_soft_std": 0.05293147638440132, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.6205596923828125, "reward_total_composite_std": 0.19636744260787964} {"timestamp_utc": "2026-04-13T00:47:02Z", "mode": "train", "global_step": 1013, "epoch": 0.1017579105976896, "loss": 0.11, "grad_norm": 12.719523429870605, "learning_rate": 6.9333333333333344e-06, "num_tokens": 1823458.0, "completions/mean_length": 48.375, "completions/min_length": 43.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.375, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9776272773742676, "rewards/meter/std": 0.017957856878638268, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8284852504730225, "rewards/repeat_soft/std": 0.12831933796405792, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.12906257808208466, "rewards/total_composite/mean": 0.6699163913726807, "rewards/total_composite/std": 0.2747206687927246, "reward": 0.6699163913726807, "reward_std": 0.2747206687927246, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1221446767449379, "sampling/sampling_logp_difference/max": 1.9013829231262207, "sampling/importance_sampling_ratio/min": 0.2517852783203125, "sampling/importance_sampling_ratio/mean": 1.0096983909606934, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6727172583341599, "clip_ratio/low_mean": 0.004032257944345474, "clip_ratio/low_min": 0.004032257944345474, "clip_ratio/high_mean": 0.10096169728785753, "clip_ratio/high_max": 0.10096169728785753, "clip_ratio/region_mean": 0.104993955232203, "reward_total_mean": 0.6699163913726807, "reward_meter_mean": 0.9776272773742676, "reward_meter_std": 0.017957856878638268, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8284852504730225, "reward_repeat_soft_std": 0.12831933796405792, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.12906257808208466, "reward_total_composite_mean": 0.6699163913726807, "reward_total_composite_std": 0.2747206687927246} {"timestamp_utc": "2026-04-13T00:47:10Z", "mode": "train", "global_step": 1014, "epoch": 0.1018583626318433, "loss": -0.0193, "grad_norm": 9.174875259399414, "learning_rate": 6.930303030303031e-06, "num_tokens": 1825505.0, "completions/mean_length": 62.875, "completions/min_length": 51.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.875, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9810818433761597, "rewards/meter/std": 0.0051107886247336864, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.696761429309845, "rewards/repeat_soft/std": 0.1879369467496872, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7444130182266235, "rewards/total_composite/std": 0.03256751969456673, "reward": 0.7444130182266235, "reward_std": 0.03256751969456673, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1133853867650032, "sampling/sampling_logp_difference/max": 1.3786158561706543, "sampling/importance_sampling_ratio/min": 0.2582221031188965, "sampling/importance_sampling_ratio/mean": 1.0160878896713257, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7588724493980408, "clip_ratio/low_mean": 0.02911993023008108, "clip_ratio/low_min": 0.02911993023008108, "clip_ratio/high_mean": 0.068996089277789, "clip_ratio/high_max": 0.068996089277789, "clip_ratio/region_mean": 0.09811601950787008, "reward_total_mean": 0.7444130182266235, "reward_meter_mean": 0.9810818433761597, "reward_meter_std": 0.0051107886247336864, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.696761429309845, "reward_repeat_soft_std": 0.1879369467496872, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7444130182266235, "reward_total_composite_std": 0.03256751969456673} {"timestamp_utc": "2026-04-13T00:47:18Z", "mode": "train", "global_step": 1015, "epoch": 0.10195881466599699, "loss": -0.0566, "grad_norm": 6.5041069984436035, "learning_rate": 6.927272727272728e-06, "num_tokens": 1827752.0, "completions/mean_length": 85.875, "completions/min_length": 67.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.875, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.9929333329200745, "rewards/meter/std": 0.0013095466420054436, "rewards/count_adherence/mean": 0.7708333134651184, "rewards/count_adherence/std": 0.0862581729888916, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.5138925313949585, "rewards/repeat_soft/std": 0.09467624872922897, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.710584282875061, "rewards/total_composite/std": 0.04534953832626343, "reward": 0.710584282875061, "reward_std": 0.04534953460097313, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0726950392127037, "sampling/sampling_logp_difference/max": 1.606274127960205, "sampling/importance_sampling_ratio/min": 0.20063376426696777, "sampling/importance_sampling_ratio/mean": 1.0150084495544434, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5652054212987423, "clip_ratio/low_mean": 0.04093753034248948, "clip_ratio/low_min": 0.04093753034248948, "clip_ratio/high_mean": 0.04083356377668679, "clip_ratio/high_max": 0.04083356377668679, "clip_ratio/region_mean": 0.08177109411917627, "reward_total_mean": 0.710584282875061, "reward_meter_mean": 0.9929333329200745, "reward_meter_std": 0.0013095466420054436, "reward_count_adherence_mean": 0.7708333134651184, "reward_count_adherence_std": 0.0862581729888916, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.5138925313949585, "reward_repeat_soft_std": 0.09467624872922897, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.710584282875061, "reward_total_composite_std": 0.04534953832626343} {"timestamp_utc": "2026-04-13T00:47:26Z", "mode": "train", "global_step": 1016, "epoch": 0.10205926670015068, "loss": 0.0928, "grad_norm": 23.87040138244629, "learning_rate": 6.9242424242424245e-06, "num_tokens": 1829261.0, "completions/mean_length": 37.625, "completions/min_length": 31.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.625, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.8352368474006653, "rewards/meter/std": 0.31732046604156494, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9436004757881165, "rewards/repeat_soft/std": 0.08932552486658096, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7495916485786438, "rewards/total_composite/std": 0.13894125819206238, "reward": 0.7495916485786438, "reward_std": 0.13894125819206238, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1702282726764679, "sampling/sampling_logp_difference/max": 1.8090128898620605, "sampling/importance_sampling_ratio/min": 0.16381575167179108, "sampling/importance_sampling_ratio/mean": 1.0048935413360596, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.032070592045784, "clip_ratio/low_mean": 0.020348837599158287, "clip_ratio/low_min": 0.020348837599158287, "clip_ratio/high_mean": 0.13843358331359923, "clip_ratio/high_max": 0.13843358331359923, "clip_ratio/region_mean": 0.15878242091275752, "reward_total_mean": 0.7495916485786438, "reward_meter_mean": 0.8352368474006653, "reward_meter_std": 0.31732046604156494, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9436004757881165, "reward_repeat_soft_std": 0.08932552486658096, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7495916485786438, "reward_total_composite_std": 0.13894125819206238} {"timestamp_utc": "2026-04-13T00:47:34Z", "mode": "train", "global_step": 1017, "epoch": 0.10215971873430436, "loss": -0.0073, "grad_norm": 24.166967391967773, "learning_rate": 6.921212121212122e-06, "num_tokens": 1830760.0, "completions/mean_length": 32.375, "completions/min_length": 24.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.375, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.8503269553184509, "rewards/meter/std": 0.2673214375972748, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9115896224975586, "rewards/repeat_soft/std": 0.06189770996570587, "rewards/judge_quality/mean": 0.5024999976158142, "rewards/judge_quality/std": 0.1440981924533844, "rewards/total_composite/mean": 0.689920961856842, "rewards/total_composite/std": 0.2935451567173004, "reward": 0.689920961856842, "reward_std": 0.2935451567173004, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1692439764738083, "sampling/sampling_logp_difference/max": 1.7939610481262207, "sampling/importance_sampling_ratio/min": 0.16630014777183533, "sampling/importance_sampling_ratio/mean": 0.9918530583381653, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.972757138311863, "clip_ratio/low_mean": 0.04285714402794838, "clip_ratio/low_min": 0.04285714402794838, "clip_ratio/high_mean": 0.1428810991346836, "clip_ratio/high_max": 0.1428810991346836, "clip_ratio/region_mean": 0.185738243162632, "reward_total_mean": 0.689920961856842, "reward_meter_mean": 0.8503269553184509, "reward_meter_std": 0.2673214375972748, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9115896224975586, "reward_repeat_soft_std": 0.06189770996570587, "reward_judge_quality_mean": 0.5024999976158142, "reward_judge_quality_std": 0.1440981924533844, "reward_total_composite_mean": 0.689920961856842, "reward_total_composite_std": 0.2935451567173004} {"timestamp_utc": "2026-04-13T00:47:41Z", "mode": "train", "global_step": 1018, "epoch": 0.10226017076845806, "loss": 0.0087, "grad_norm": 10.076597213745117, "learning_rate": 6.918181818181818e-06, "num_tokens": 1832606.0, "completions/mean_length": 59.75, "completions/min_length": 51.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.75, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9898721575737, "rewards/meter/std": 0.003591838525608182, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8652453422546387, "rewards/repeat_soft/std": 0.1552790254354477, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.7475919723510742, "rewards/total_composite/std": 0.04369426518678665, "reward": 0.7475919723510742, "reward_std": 0.04369427636265755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13600337505340576, "sampling/sampling_logp_difference/max": 1.7163143157958984, "sampling/importance_sampling_ratio/min": 0.17972734570503235, "sampling/importance_sampling_ratio/mean": 1.0280085802078247, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0347401946783066, "clip_ratio/low_mean": 0.04063659766688943, "clip_ratio/low_min": 0.04063659766688943, "clip_ratio/high_mean": 0.058341961819678545, "clip_ratio/high_max": 0.058341961819678545, "clip_ratio/region_mean": 0.09897855948656797, "reward_total_mean": 0.7475919723510742, "reward_meter_mean": 0.9898721575737, "reward_meter_std": 0.003591838525608182, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8652453422546387, "reward_repeat_soft_std": 0.1552790254354477, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.7475919723510742, "reward_total_composite_std": 0.04369426518678665} {"timestamp_utc": "2026-04-13T00:47:50Z", "mode": "train", "global_step": 1019, "epoch": 0.10236062280261175, "loss": -0.0012, "grad_norm": 12.608658790588379, "learning_rate": 6.915151515151515e-06, "num_tokens": 1834447.0, "completions/mean_length": 63.125, "completions/min_length": 47.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.125, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.38824620842933655, "rewards/meter/std": 0.3917814791202545, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9226629734039307, "rewards/repeat_soft/std": 0.08499263226985931, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.49910208582878113, "rewards/total_composite/std": 0.17882928252220154, "reward": 0.49910208582878113, "reward_std": 0.17882928252220154, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1863686740398407, "sampling/sampling_logp_difference/max": 1.7371406555175781, "sampling/importance_sampling_ratio/min": 0.17602300643920898, "sampling/importance_sampling_ratio/mean": 1.0096684694290161, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.73690564930439, "clip_ratio/low_mean": 0.13342138938605785, "clip_ratio/low_min": 0.13342138938605785, "clip_ratio/high_mean": 0.07992042042315006, "clip_ratio/high_max": 0.07992042042315006, "clip_ratio/region_mean": 0.21334180980920792, "reward_total_mean": 0.49910208582878113, "reward_meter_mean": 0.38824620842933655, "reward_meter_std": 0.3917814791202545, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9226629734039307, "reward_repeat_soft_std": 0.08499263226985931, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.49910208582878113, "reward_total_composite_std": 0.17882928252220154} {"timestamp_utc": "2026-04-13T00:47:57Z", "mode": "train", "global_step": 1020, "epoch": 0.10246107483676545, "loss": -0.0292, "grad_norm": 6.374295711517334, "learning_rate": 6.912121212121212e-06, "num_tokens": 1836609.0, "completions/mean_length": 91.25, "completions/min_length": 82.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.25, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.9905205965042114, "rewards/meter/std": 0.0013658568495884538, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.6730790138244629, "rewards/repeat_soft/std": 0.15986157953739166, "rewards/judge_quality/mean": 0.2887499928474426, "rewards/judge_quality/std": 0.11630470305681229, "rewards/total_composite/mean": 0.6422610282897949, "rewards/total_composite/std": 0.2616559863090515, "reward": 0.6422610282897949, "reward_std": 0.2616559863090515, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09636439383029938, "sampling/sampling_logp_difference/max": 2.429431438446045, "sampling/importance_sampling_ratio/min": 0.08808690309524536, "sampling/importance_sampling_ratio/mean": 1.0163277387619019, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7497988119721413, "clip_ratio/low_mean": 0.007621951401233673, "clip_ratio/low_min": 0.007621951401233673, "clip_ratio/high_mean": 0.09133840352296829, "clip_ratio/high_max": 0.09133840352296829, "clip_ratio/region_mean": 0.09896035492420197, "reward_total_mean": 0.6422610282897949, "reward_meter_mean": 0.9905205965042114, "reward_meter_std": 0.0013658568495884538, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.6730790138244629, "reward_repeat_soft_std": 0.15986157953739166, "reward_judge_quality_mean": 0.2887499928474426, "reward_judge_quality_std": 0.11630470305681229, "reward_total_composite_mean": 0.6422610282897949, "reward_total_composite_std": 0.2616559863090515} {"timestamp_utc": "2026-04-13T00:48:05Z", "mode": "train", "global_step": 1021, "epoch": 0.10256152687091914, "loss": 0.093, "grad_norm": 11.945598602294922, "learning_rate": 6.90909090909091e-06, "num_tokens": 1838705.0, "completions/mean_length": 70.0, "completions/min_length": 60.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.0, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.9483002424240112, "rewards/meter/std": 0.1070336401462555, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9001034498214722, "rewards/repeat_soft/std": 0.09708748757839203, "rewards/judge_quality/mean": 0.3137499988079071, "rewards/judge_quality/std": 0.12772038578987122, "rewards/total_composite/mean": 0.7233704328536987, "rewards/total_composite/std": 0.0720343142747879, "reward": 0.7233704328536987, "reward_std": 0.0720343142747879, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12920977175235748, "sampling/sampling_logp_difference/max": 1.3941001892089844, "sampling/importance_sampling_ratio/min": 0.24805612862110138, "sampling/importance_sampling_ratio/mean": 1.0157513618469238, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9031215980648994, "clip_ratio/low_mean": 0.039332279935479164, "clip_ratio/low_min": 0.039332279935479164, "clip_ratio/high_mean": 0.0986167024821043, "clip_ratio/high_max": 0.0986167024821043, "clip_ratio/region_mean": 0.13794898241758347, "reward_total_mean": 0.7233704328536987, "reward_meter_mean": 0.9483002424240112, "reward_meter_std": 0.1070336401462555, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9001034498214722, "reward_repeat_soft_std": 0.09708748757839203, "reward_judge_quality_mean": 0.3137499988079071, "reward_judge_quality_std": 0.12772038578987122, "reward_total_composite_mean": 0.7233704328536987, "reward_total_composite_std": 0.0720343142747879} {"timestamp_utc": "2026-04-13T00:48:12Z", "mode": "train", "global_step": 1022, "epoch": 0.10266197890507282, "loss": 0.0143, "grad_norm": 14.98851203918457, "learning_rate": 6.906060606060606e-06, "num_tokens": 1840266.0, "completions/mean_length": 45.125, "completions/min_length": 37.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.125, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.9875601530075073, "rewards/meter/std": 0.011146973818540573, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9685879945755005, "rewards/repeat_soft/std": 0.012200173921883106, "rewards/judge_quality/mean": 0.38874998688697815, "rewards/judge_quality/std": 0.08675704896450043, "rewards/total_composite/mean": 0.8078858852386475, "rewards/total_composite/std": 0.025832535699009895, "reward": 0.8078858852386475, "reward_std": 0.0258325282484293, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1324063390493393, "sampling/sampling_logp_difference/max": 1.4377503395080566, "sampling/importance_sampling_ratio/min": 0.23746135830879211, "sampling/importance_sampling_ratio/mean": 1.0283337831497192, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.982489638030529, "clip_ratio/low_mean": 0.017173423897475004, "clip_ratio/low_min": 0.017173423897475004, "clip_ratio/high_mean": 0.09047915320843458, "clip_ratio/high_max": 0.09047915320843458, "clip_ratio/region_mean": 0.10765257710590959, "reward_total_mean": 0.8078858852386475, "reward_meter_mean": 0.9875601530075073, "reward_meter_std": 0.011146973818540573, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9685879945755005, "reward_repeat_soft_std": 0.012200173921883106, "reward_judge_quality_mean": 0.38874998688697815, "reward_judge_quality_std": 0.08675704896450043, "reward_total_composite_mean": 0.8078858852386475, "reward_total_composite_std": 0.025832535699009895} {"timestamp_utc": "2026-04-13T00:48:24Z", "mode": "train", "global_step": 1023, "epoch": 0.10276243093922652, "loss": -0.1034, "grad_norm": 2.8847508430480957, "learning_rate": 6.903030303030304e-06, "num_tokens": 1841779.0, "completions/mean_length": 94.125, "completions/min_length": 30.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 34.42857360839844, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.7452960014343262, "rewards/meter/std": 0.3481470048427582, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9724948406219482, "rewards/repeat_soft/std": 0.029870249330997467, "rewards/judge_quality/mean": 0.41874998807907104, "rewards/judge_quality/std": 0.18074746429920197, "rewards/total_composite/mean": 0.6701710224151611, "rewards/total_composite/std": 0.29481446743011475, "reward": 0.6701710224151611, "reward_std": 0.29481446743011475, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14299020171165466, "sampling/sampling_logp_difference/max": 0.8285234570503235, "sampling/importance_sampling_ratio/min": 0.43669363856315613, "sampling/importance_sampling_ratio/mean": 1.0270124673843384, "sampling/importance_sampling_ratio/max": 1.9636256694793701, "entropy": 0.8727670237421989, "clip_ratio/low_mean": 0.013888888992369175, "clip_ratio/low_min": 0.013888888992369175, "clip_ratio/high_mean": 0.11069969926029444, "clip_ratio/high_max": 0.11069969926029444, "clip_ratio/region_mean": 0.12458858825266361, "reward_total_mean": 0.6701710224151611, "reward_meter_mean": 0.7452960014343262, "reward_meter_std": 0.3481470048427582, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9724948406219482, "reward_repeat_soft_std": 0.029870249330997467, "reward_judge_quality_mean": 0.41874998807907104, "reward_judge_quality_std": 0.18074746429920197, "reward_total_composite_mean": 0.6701710224151611, "reward_total_composite_std": 0.29481446743011475} {"timestamp_utc": "2026-04-13T00:48:31Z", "mode": "train", "global_step": 1024, "epoch": 0.10286288297338021, "loss": -0.0049, "grad_norm": 19.75912094116211, "learning_rate": 6.9e-06, "num_tokens": 1843108.0, "completions/mean_length": 20.125, "completions/min_length": 17.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.125, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.9907851219177246, "rewards/meter/std": 0.005948507227003574, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9386930465698242, "rewards/repeat_soft/std": 0.04203351214528084, "rewards/judge_quality/mean": 0.39249998331069946, "rewards/judge_quality/std": 0.08892211318016052, "rewards/total_composite/mean": 0.8074725866317749, "rewards/total_composite/std": 0.02888883836567402, "reward": 0.8074725866317749, "reward_std": 0.028888842090964317, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15156129002571106, "sampling/sampling_logp_difference/max": 1.9275580644607544, "sampling/importance_sampling_ratio/min": 0.14550307393074036, "sampling/importance_sampling_ratio/mean": 1.01750910282135, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7725184187293053, "clip_ratio/low_mean": 0.03670634934678674, "clip_ratio/low_min": 0.03670634934678674, "clip_ratio/high_mean": 0.09252347331494093, "clip_ratio/high_max": 0.09252347331494093, "clip_ratio/region_mean": 0.12922982266172767, "reward_total_mean": 0.8074725866317749, "reward_meter_mean": 0.9907851219177246, "reward_meter_std": 0.005948507227003574, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9386930465698242, "reward_repeat_soft_std": 0.04203351214528084, "reward_judge_quality_mean": 0.39249998331069946, "reward_judge_quality_std": 0.08892211318016052, "reward_total_composite_mean": 0.8074725866317749, "reward_total_composite_std": 0.02888883836567402} {"timestamp_utc": "2026-04-13T00:48:44Z", "mode": "train", "global_step": 1025, "epoch": 0.10296333500753391, "loss": -0.105, "grad_norm": 3.863487958908081, "learning_rate": 6.896969696969697e-06, "num_tokens": 1844627.0, "completions/mean_length": 101.875, "completions/min_length": 38.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 43.28571701049805, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.5059927701950073, "rewards/meter/std": 0.4107321500778198, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.991308331489563, "rewards/repeat_soft/std": 0.01502656564116478, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.13452960550785065, "rewards/total_composite/mean": 0.5580775737762451, "rewards/total_composite/std": 0.27474841475486755, "reward": 0.5580775737762451, "reward_std": 0.27474841475486755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13738718628883362, "sampling/sampling_logp_difference/max": 0.9412946701049805, "sampling/importance_sampling_ratio/min": 0.3901224434375763, "sampling/importance_sampling_ratio/mean": 0.99459308385849, "sampling/importance_sampling_ratio/max": 1.7453088760375977, "entropy": 0.965722344815731, "clip_ratio/low_mean": 0.026815409772098064, "clip_ratio/low_min": 0.026815409772098064, "clip_ratio/high_mean": 0.11470973119139671, "clip_ratio/high_max": 0.11470973119139671, "clip_ratio/region_mean": 0.14152514096349478, "reward_total_mean": 0.5580775737762451, "reward_meter_mean": 0.5059927701950073, "reward_meter_std": 0.4107321500778198, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.991308331489563, "reward_repeat_soft_std": 0.01502656564116478, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.13452960550785065, "reward_total_composite_mean": 0.5580775737762451, "reward_total_composite_std": 0.27474841475486755} {"timestamp_utc": "2026-04-13T00:48:51Z", "mode": "train", "global_step": 1026, "epoch": 0.10306378704168759, "loss": -0.1164, "grad_norm": 15.671709060668945, "learning_rate": 6.893939393939395e-06, "num_tokens": 1845850.0, "completions/mean_length": 26.875, "completions/min_length": 19.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.875, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9756706357002258, "rewards/meter/std": 0.03890342637896538, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8900974988937378, "rewards/repeat_soft/std": 0.10846816748380661, "rewards/judge_quality/mean": 0.2837499976158142, "rewards/judge_quality/std": 0.136165589094162, "rewards/total_composite/mean": 0.7631865739822388, "rewards/total_composite/std": 0.043742138892412186, "reward": 0.7631865739822388, "reward_std": 0.04374213516712189, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13474354147911072, "sampling/sampling_logp_difference/max": 1.320474624633789, "sampling/importance_sampling_ratio/min": 0.26700854301452637, "sampling/importance_sampling_ratio/mean": 1.0047478675842285, "sampling/importance_sampling_ratio/max": 1.8857313394546509, "entropy": 1.2420333176851273, "clip_ratio/low_mean": 0.04709308035671711, "clip_ratio/low_min": 0.04709308035671711, "clip_ratio/high_mean": 0.04411300644278526, "clip_ratio/high_max": 0.04411300644278526, "clip_ratio/region_mean": 0.09120608679950237, "reward_total_mean": 0.7631865739822388, "reward_meter_mean": 0.9756706357002258, "reward_meter_std": 0.03890342637896538, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8900974988937378, "reward_repeat_soft_std": 0.10846816748380661, "reward_judge_quality_mean": 0.2837499976158142, "reward_judge_quality_std": 0.136165589094162, "reward_total_composite_mean": 0.7631865739822388, "reward_total_composite_std": 0.043742138892412186} {"timestamp_utc": "2026-04-13T00:48:59Z", "mode": "train", "global_step": 1027, "epoch": 0.10316423907584128, "loss": 0.0807, "grad_norm": 19.63422966003418, "learning_rate": 6.890909090909092e-06, "num_tokens": 1848056.0, "completions/mean_length": 86.75, "completions/min_length": 79.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.75, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.8377218842506409, "rewards/meter/std": 0.28340837359428406, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9672204256057739, "rewards/repeat_soft/std": 0.02763226255774498, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7246968746185303, "rewards/total_composite/std": 0.1286119818687439, "reward": 0.7246968746185303, "reward_std": 0.1286119669675827, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19207976758480072, "sampling/sampling_logp_difference/max": 1.6206269264221191, "sampling/importance_sampling_ratio/min": 0.19777466356754303, "sampling/importance_sampling_ratio/mean": 1.0198649168014526, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9157451689243317, "clip_ratio/low_mean": 0.014473684132099152, "clip_ratio/low_min": 0.014473684132099152, "clip_ratio/high_mean": 0.15463581215590239, "clip_ratio/high_max": 0.15463581215590239, "clip_ratio/region_mean": 0.16910949628800154, "reward_total_mean": 0.7246968746185303, "reward_meter_mean": 0.8377218842506409, "reward_meter_std": 0.28340837359428406, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9672204256057739, "reward_repeat_soft_std": 0.02763226255774498, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7246968746185303, "reward_total_composite_std": 0.1286119818687439} {"timestamp_utc": "2026-04-13T00:49:07Z", "mode": "train", "global_step": 1028, "epoch": 0.10326469110999498, "loss": -0.0119, "grad_norm": 12.319024085998535, "learning_rate": 6.887878787878789e-06, "num_tokens": 1849820.0, "completions/mean_length": 37.5, "completions/min_length": 34.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.5, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9890105724334717, "rewards/meter/std": 0.004029208794236183, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9374485015869141, "rewards/repeat_soft/std": 0.04705830663442612, "rewards/judge_quality/mean": 0.5012500286102295, "rewards/judge_quality/std": 0.16974246501922607, "rewards/total_composite/mean": 0.7891746163368225, "rewards/total_composite/std": 0.05050061643123627, "reward": 0.7891746163368225, "reward_std": 0.05050060898065567, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14430801570415497, "sampling/sampling_logp_difference/max": 2.3059041500091553, "sampling/importance_sampling_ratio/min": 0.09966864436864853, "sampling/importance_sampling_ratio/mean": 0.9839019775390625, "sampling/importance_sampling_ratio/max": 1.9462635517120361, "entropy": 0.8270372077822685, "clip_ratio/low_mean": 0.09882117947563529, "clip_ratio/low_min": 0.09882117947563529, "clip_ratio/high_mean": 0.015243902802467346, "clip_ratio/high_max": 0.015243902802467346, "clip_ratio/region_mean": 0.11406508227810264, "reward_total_mean": 0.7891746163368225, "reward_meter_mean": 0.9890105724334717, "reward_meter_std": 0.004029208794236183, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9374485015869141, "reward_repeat_soft_std": 0.04705830663442612, "reward_judge_quality_mean": 0.5012500286102295, "reward_judge_quality_std": 0.16974246501922607, "reward_total_composite_mean": 0.7891746163368225, "reward_total_composite_std": 0.05050061643123627} {"timestamp_utc": "2026-04-13T00:49:15Z", "mode": "train", "global_step": 1029, "epoch": 0.10336514314414867, "loss": -0.0208, "grad_norm": 10.91002082824707, "learning_rate": 6.8848484848484854e-06, "num_tokens": 1851717.0, "completions/mean_length": 62.125, "completions/min_length": 51.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.125, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.9917694330215454, "rewards/meter/std": 0.0015094189438968897, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9273158311843872, "rewards/repeat_soft/std": 0.04798102751374245, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7711528539657593, "rewards/total_composite/std": 0.019537733867764473, "reward": 0.7711528539657593, "reward_std": 0.01953774504363537, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13854002952575684, "sampling/sampling_logp_difference/max": 2.1248269081115723, "sampling/importance_sampling_ratio/min": 0.11945364624261856, "sampling/importance_sampling_ratio/mean": 0.993834376335144, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9462441578507423, "clip_ratio/low_mean": 0.016826923936605453, "clip_ratio/low_min": 0.016826923936605453, "clip_ratio/high_mean": 0.12894496275112033, "clip_ratio/high_max": 0.12894496275112033, "clip_ratio/region_mean": 0.14577188668772578, "reward_total_mean": 0.7711528539657593, "reward_meter_mean": 0.9917694330215454, "reward_meter_std": 0.0015094189438968897, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9273158311843872, "reward_repeat_soft_std": 0.04798102751374245, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7711528539657593, "reward_total_composite_std": 0.019537733867764473} {"timestamp_utc": "2026-04-13T00:49:22Z", "mode": "train", "global_step": 1030, "epoch": 0.10346559517830237, "loss": 0.0296, "grad_norm": 16.42207145690918, "learning_rate": 6.881818181818183e-06, "num_tokens": 1853294.0, "completions/mean_length": 39.125, "completions/min_length": 33.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.125, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.5254687666893005, "rewards/meter/std": 0.47569042444229126, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9317286610603333, "rewards/repeat_soft/std": 0.03859012946486473, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.25587037205696106, "rewards/total_composite/mean": 0.6630088090896606, "rewards/total_composite/std": 0.20491495728492737, "reward": 0.6630088090896606, "reward_std": 0.20491494238376617, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14891985058784485, "sampling/sampling_logp_difference/max": 2.27618145942688, "sampling/importance_sampling_ratio/min": 0.10267552733421326, "sampling/importance_sampling_ratio/mean": 1.0148261785507202, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7918422818183899, "clip_ratio/low_mean": 0.06058131763711572, "clip_ratio/low_min": 0.06058131763711572, "clip_ratio/high_mean": 0.07235624454915524, "clip_ratio/high_max": 0.07235624454915524, "clip_ratio/region_mean": 0.13293756218627095, "reward_total_mean": 0.6630088090896606, "reward_meter_mean": 0.5254687666893005, "reward_meter_std": 0.47569042444229126, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9317286610603333, "reward_repeat_soft_std": 0.03859012946486473, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.25587037205696106, "reward_total_composite_mean": 0.6630088090896606, "reward_total_composite_std": 0.20491495728492737} {"timestamp_utc": "2026-04-13T00:49:30Z", "mode": "train", "global_step": 1031, "epoch": 0.10356604721245605, "loss": -0.0615, "grad_norm": 8.377963066101074, "learning_rate": 6.878787878787879e-06, "num_tokens": 1855202.0, "completions/mean_length": 82.5, "completions/min_length": 67.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.5, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.9563207626342773, "rewards/meter/std": 0.10445500910282135, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8882862329483032, "rewards/repeat_soft/std": 0.09980380535125732, "rewards/judge_quality/mean": 0.42625001072883606, "rewards/judge_quality/std": 0.14647647738456726, "rewards/total_composite/mean": 0.7595479488372803, "rewards/total_composite/std": 0.06623243540525436, "reward": 0.7595479488372803, "reward_std": 0.06623242795467377, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12891200184822083, "sampling/sampling_logp_difference/max": 1.518151044845581, "sampling/importance_sampling_ratio/min": 0.24388617277145386, "sampling/importance_sampling_ratio/mean": 1.016065239906311, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0067602843046188, "clip_ratio/low_mean": 0.037979744374752045, "clip_ratio/low_min": 0.037979744374752045, "clip_ratio/high_mean": 0.0724629731848836, "clip_ratio/high_max": 0.0724629731848836, "clip_ratio/region_mean": 0.11044271755963564, "reward_total_mean": 0.7595479488372803, "reward_meter_mean": 0.9563207626342773, "reward_meter_std": 0.10445500910282135, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8882862329483032, "reward_repeat_soft_std": 0.09980380535125732, "reward_judge_quality_mean": 0.42625001072883606, "reward_judge_quality_std": 0.14647647738456726, "reward_total_composite_mean": 0.7595479488372803, "reward_total_composite_std": 0.06623243540525436} {"timestamp_utc": "2026-04-13T00:49:38Z", "mode": "train", "global_step": 1032, "epoch": 0.10366649924660974, "loss": 0.0872, "grad_norm": 13.32891845703125, "learning_rate": 6.875757575757576e-06, "num_tokens": 1857115.0, "completions/mean_length": 63.125, "completions/min_length": 57.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.125, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9915275573730469, "rewards/meter/std": 0.003440647153183818, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9658991098403931, "rewards/repeat_soft/std": 0.04060649499297142, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.7782773375511169, "rewards/total_composite/std": 0.02043924108147621, "reward": 0.7782773375511169, "reward_std": 0.020439239218831062, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15546822547912598, "sampling/sampling_logp_difference/max": 2.0468969345092773, "sampling/importance_sampling_ratio/min": 0.12913499772548676, "sampling/importance_sampling_ratio/mean": 1.0230910778045654, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.024569258093834, "clip_ratio/low_mean": 0.019039166159927845, "clip_ratio/low_min": 0.019039166159927845, "clip_ratio/high_mean": 0.08530076127499342, "clip_ratio/high_max": 0.08530076127499342, "clip_ratio/region_mean": 0.10433992743492126, "reward_total_mean": 0.7782773375511169, "reward_meter_mean": 0.9915275573730469, "reward_meter_std": 0.003440647153183818, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9658991098403931, "reward_repeat_soft_std": 0.04060649499297142, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.7782773375511169, "reward_total_composite_std": 0.02043924108147621} {"timestamp_utc": "2026-04-13T00:49:50Z", "mode": "train", "global_step": 1033, "epoch": 0.10376695128076344, "loss": -0.0091, "grad_norm": 6.469017028808594, "learning_rate": 6.872727272727273e-06, "num_tokens": 1858387.0, "completions/mean_length": 85.0, "completions/min_length": 19.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 24.000001907348633, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.8123348355293274, "rewards/meter/std": 0.31077060103416443, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9655071496963501, "rewards/repeat_soft/std": 0.01375842746347189, "rewards/judge_quality/mean": 0.3499999940395355, "rewards/judge_quality/std": 0.1511857807636261, "rewards/total_composite/mean": 0.7171014547348022, "rewards/total_composite/std": 0.14602714776992798, "reward": 0.7171014547348022, "reward_std": 0.14602716267108917, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14411786198616028, "sampling/sampling_logp_difference/max": 1.653977632522583, "sampling/importance_sampling_ratio/min": 0.19128751754760742, "sampling/importance_sampling_ratio/mean": 1.029304027557373, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.986639603972435, "clip_ratio/low_mean": 0.03173913061618805, "clip_ratio/low_min": 0.03173913061618805, "clip_ratio/high_mean": 0.07809332152828574, "clip_ratio/high_max": 0.07809332152828574, "clip_ratio/region_mean": 0.10983245214447379, "reward_total_mean": 0.7171014547348022, "reward_meter_mean": 0.8123348355293274, "reward_meter_std": 0.31077060103416443, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9655071496963501, "reward_repeat_soft_std": 0.01375842746347189, "reward_judge_quality_mean": 0.3499999940395355, "reward_judge_quality_std": 0.1511857807636261, "reward_total_composite_mean": 0.7171014547348022, "reward_total_composite_std": 0.14602714776992798} {"timestamp_utc": "2026-04-13T00:50:01Z", "mode": "train", "global_step": 1034, "epoch": 0.10386740331491713, "loss": 0.0345, "grad_norm": 9.0579252243042, "learning_rate": 6.869696969696971e-06, "num_tokens": 1860300.0, "completions/mean_length": 58.125, "completions/min_length": 49.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.125, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9914954304695129, "rewards/meter/std": 0.008340935222804546, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9745539426803589, "rewards/repeat_soft/std": 0.025256892666220665, "rewards/judge_quality/mean": 0.5637500286102295, "rewards/judge_quality/std": 0.22012579441070557, "rewards/total_composite/mean": 0.8627533316612244, "rewards/total_composite/std": 0.0641782134771347, "reward": 0.8627533316612244, "reward_std": 0.0641782283782959, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13077011704444885, "sampling/sampling_logp_difference/max": 2.1905269622802734, "sampling/importance_sampling_ratio/min": 0.11185778677463531, "sampling/importance_sampling_ratio/mean": 1.0189530849456787, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8985662311315536, "clip_ratio/low_mean": 0.09033840615302324, "clip_ratio/low_min": 0.09033840615302324, "clip_ratio/high_mean": 0.024396845139563084, "clip_ratio/high_max": 0.024396845139563084, "clip_ratio/region_mean": 0.11473525129258633, "reward_total_mean": 0.8627533316612244, "reward_meter_mean": 0.9914954304695129, "reward_meter_std": 0.008340935222804546, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9745539426803589, "reward_repeat_soft_std": 0.025256892666220665, "reward_judge_quality_mean": 0.5637500286102295, "reward_judge_quality_std": 0.22012579441070557, "reward_total_composite_mean": 0.8627533316612244, "reward_total_composite_std": 0.0641782134771347} {"timestamp_utc": "2026-04-13T00:50:10Z", "mode": "train", "global_step": 1035, "epoch": 0.10396785534907081, "loss": 0.0234, "grad_norm": 7.149699687957764, "learning_rate": 6.866666666666667e-06, "num_tokens": 1862695.0, "completions/mean_length": 104.375, "completions/min_length": 92.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.375, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.9042094349861145, "rewards/meter/std": 0.2337023913860321, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8690142631530762, "rewards/repeat_soft/std": 0.08869493752717972, "rewards/judge_quality/mean": 0.29249998927116394, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.5944747924804688, "rewards/total_composite/std": 0.26583969593048096, "reward": 0.5944747924804688, "reward_std": 0.26583969593048096, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1001431941986084, "sampling/sampling_logp_difference/max": 0.9916791915893555, "sampling/importance_sampling_ratio/min": 0.3709532916545868, "sampling/importance_sampling_ratio/mean": 1.0340591669082642, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9305148720741272, "clip_ratio/low_mean": 0.03611966222524643, "clip_ratio/low_min": 0.03611966222524643, "clip_ratio/high_mean": 0.08341884799301624, "clip_ratio/high_max": 0.08341884799301624, "clip_ratio/region_mean": 0.11953851021826267, "reward_total_mean": 0.5944747924804688, "reward_meter_mean": 0.9042094349861145, "reward_meter_std": 0.2337023913860321, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8690142631530762, "reward_repeat_soft_std": 0.08869493752717972, "reward_judge_quality_mean": 0.29249998927116394, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.5944747924804688, "reward_total_composite_std": 0.26583969593048096} {"timestamp_utc": "2026-04-13T00:50:18Z", "mode": "train", "global_step": 1036, "epoch": 0.1040683073832245, "loss": -0.0636, "grad_norm": 7.140467166900635, "learning_rate": 6.8636363636363645e-06, "num_tokens": 1864643.0, "completions/mean_length": 65.5, "completions/min_length": 51.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.5, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9769264459609985, "rewards/meter/std": 0.03486189991235733, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8884038329124451, "rewards/repeat_soft/std": 0.14198140799999237, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.21357084810733795, "rewards/total_composite/mean": 0.7055364847183228, "rewards/total_composite/std": 0.29068565368652344, "reward": 0.7055364847183228, "reward_std": 0.29068565368652344, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1317613273859024, "sampling/sampling_logp_difference/max": 2.0092687606811523, "sampling/importance_sampling_ratio/min": 0.13408668339252472, "sampling/importance_sampling_ratio/mean": 0.9972166419029236, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.744259424507618, "clip_ratio/low_mean": 0.014705882407724857, "clip_ratio/low_min": 0.014705882407724857, "clip_ratio/high_mean": 0.11889945156872272, "clip_ratio/high_max": 0.11889945156872272, "clip_ratio/region_mean": 0.13360533397644758, "reward_total_mean": 0.7055364847183228, "reward_meter_mean": 0.9769264459609985, "reward_meter_std": 0.03486189991235733, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8884038329124451, "reward_repeat_soft_std": 0.14198140799999237, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.21357084810733795, "reward_total_composite_mean": 0.7055364847183228, "reward_total_composite_std": 0.29068565368652344} {"timestamp_utc": "2026-04-13T00:50:26Z", "mode": "train", "global_step": 1037, "epoch": 0.1041687594173782, "loss": 0.0089, "grad_norm": 13.363117218017578, "learning_rate": 6.860606060606061e-06, "num_tokens": 1866598.0, "completions/mean_length": 63.375, "completions/min_length": 59.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.375, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.538659930229187, "rewards/meter/std": 0.3776288330554962, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.976222574710846, "rewards/repeat_soft/std": 0.02054116129875183, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.6010192036628723, "rewards/total_composite/std": 0.15981775522232056, "reward": 0.6010192036628723, "reward_std": 0.15981774032115936, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14119084179401398, "sampling/sampling_logp_difference/max": 2.066161632537842, "sampling/importance_sampling_ratio/min": 0.12667106091976166, "sampling/importance_sampling_ratio/mean": 0.9948351979255676, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5988906510174274, "clip_ratio/low_mean": 0.058920823503285646, "clip_ratio/low_min": 0.058920823503285646, "clip_ratio/high_mean": 0.040390859823673964, "clip_ratio/high_max": 0.040390859823673964, "clip_ratio/region_mean": 0.09931168332695961, "reward_total_mean": 0.6010192036628723, "reward_meter_mean": 0.538659930229187, "reward_meter_std": 0.3776288330554962, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.976222574710846, "reward_repeat_soft_std": 0.02054116129875183, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.6010192036628723, "reward_total_composite_std": 0.15981775522232056} {"timestamp_utc": "2026-04-13T00:50:39Z", "mode": "train", "global_step": 1038, "epoch": 0.1042692114515319, "loss": -0.1124, "grad_norm": 2.7273805141448975, "learning_rate": 6.857575757575758e-06, "num_tokens": 1868348.0, "completions/mean_length": 98.75, "completions/min_length": 35.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 39.71428680419922, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.7626007795333862, "rewards/meter/std": 0.4051229953765869, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9568420052528381, "rewards/repeat_soft/std": 0.04375278204679489, "rewards/judge_quality/mean": 0.5562500357627869, "rewards/judge_quality/std": 0.26997023820877075, "rewards/total_composite/mean": 0.722604513168335, "rewards/total_composite/std": 0.31567829847335815, "reward": 0.722604513168335, "reward_std": 0.31567829847335815, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16622546315193176, "sampling/sampling_logp_difference/max": 1.0006532669067383, "sampling/importance_sampling_ratio/min": 0.3676392138004303, "sampling/importance_sampling_ratio/mean": 0.9896756410598755, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9509481340646744, "clip_ratio/low_mean": 0.022435897961258888, "clip_ratio/low_min": 0.022435897961258888, "clip_ratio/high_mean": 0.16803727857768536, "clip_ratio/high_max": 0.16803727857768536, "clip_ratio/region_mean": 0.19047317653894424, "reward_total_mean": 0.722604513168335, "reward_meter_mean": 0.7626007795333862, "reward_meter_std": 0.4051229953765869, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9568420052528381, "reward_repeat_soft_std": 0.04375278204679489, "reward_judge_quality_mean": 0.5562500357627869, "reward_judge_quality_std": 0.26997023820877075, "reward_total_composite_mean": 0.722604513168335, "reward_total_composite_std": 0.31567829847335815} {"timestamp_utc": "2026-04-13T00:50:52Z", "mode": "train", "global_step": 1039, "epoch": 0.10436966348568559, "loss": 0.0162, "grad_norm": 28.57527732849121, "learning_rate": 6.854545454545455e-06, "num_tokens": 1869957.0, "completions/mean_length": 32.125, "completions/min_length": 27.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.125, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.5219188928604126, "rewards/meter/std": 0.36124399304389954, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9961345791816711, "rewards/repeat_soft/std": 0.005192012991756201, "rewards/judge_quality/mean": 0.6349999904632568, "rewards/judge_quality/std": 0.3139154016971588, "rewards/total_composite/mean": 0.6749769449234009, "rewards/total_composite/std": 0.19299302995204926, "reward": 0.6749769449234009, "reward_std": 0.19299301505088806, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17350773513317108, "sampling/sampling_logp_difference/max": 1.689992070198059, "sampling/importance_sampling_ratio/min": 0.18452098965644836, "sampling/importance_sampling_ratio/mean": 1.0271848440170288, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9871536269783974, "clip_ratio/low_mean": 0.10587771050632, "clip_ratio/low_min": 0.10587771050632, "clip_ratio/high_mean": 0.05360060930252075, "clip_ratio/high_max": 0.05360060930252075, "clip_ratio/region_mean": 0.15947831980884075, "reward_total_mean": 0.6749769449234009, "reward_meter_mean": 0.5219188928604126, "reward_meter_std": 0.36124399304389954, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9961345791816711, "reward_repeat_soft_std": 0.005192012991756201, "reward_judge_quality_mean": 0.6349999904632568, "reward_judge_quality_std": 0.3139154016971588, "reward_total_composite_mean": 0.6749769449234009, "reward_total_composite_std": 0.19299302995204926} {"timestamp_utc": "2026-04-13T00:51:22Z", "mode": "train", "global_step": 1040, "epoch": 0.10447011551983927, "loss": 0.0303, "grad_norm": 13.448856353759766, "learning_rate": 6.851515151515153e-06, "num_tokens": 1871960.0, "completions/mean_length": 69.375, "completions/min_length": 65.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.375, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.7374683618545532, "rewards/meter/std": 0.3877306580543518, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9459856748580933, "rewards/repeat_soft/std": 0.05902494862675667, "rewards/judge_quality/mean": 0.20999999344348907, "rewards/judge_quality/std": 0.22449944913387299, "rewards/total_composite/mean": 0.6019593477249146, "rewards/total_composite/std": 0.15028560161590576, "reward": 0.6019593477249146, "reward_std": 0.15028560161590576, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15199393033981323, "sampling/sampling_logp_difference/max": 4.008288383483887, "sampling/importance_sampling_ratio/min": 0.018164457753300667, "sampling/importance_sampling_ratio/mean": 1.0216128826141357, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1405395716428757, "clip_ratio/low_mean": 0.06338448077440262, "clip_ratio/low_min": 0.06338448077440262, "clip_ratio/high_mean": 0.09737133421003819, "clip_ratio/high_max": 0.09737133421003819, "clip_ratio/region_mean": 0.1607558149844408, "reward_total_mean": 0.6019593477249146, "reward_meter_mean": 0.7374683618545532, "reward_meter_std": 0.3877306580543518, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9459856748580933, "reward_repeat_soft_std": 0.05902494862675667, "reward_judge_quality_mean": 0.20999999344348907, "reward_judge_quality_std": 0.22449944913387299, "reward_total_composite_mean": 0.6019593477249146, "reward_total_composite_std": 0.15028560161590576} {"timestamp_utc": "2026-04-13T00:51:37Z", "mode": "train", "global_step": 1041, "epoch": 0.10457056755399297, "loss": -0.232, "grad_norm": 1.746078610420227, "learning_rate": 6.848484848484849e-06, "num_tokens": 1874343.0, "completions/mean_length": 174.875, "completions/min_length": 105.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 126.71429443359375, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 153.0, "rewards/meter/mean": 0.8672534227371216, "rewards/meter/std": 0.34703925251960754, "rewards/count_adherence/mean": 0.6458333730697632, "rewards/count_adherence/std": 0.20773723721504211, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8143404722213745, "rewards/repeat_soft/std": 0.0835551917552948, "rewards/judge_quality/mean": 0.20875000953674316, "rewards/judge_quality/std": 0.11038083583116531, "rewards/total_composite/mean": 0.6136913299560547, "rewards/total_composite/std": 0.25135669112205505, "reward": 0.6136913299560547, "reward_std": 0.25135669112205505, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10185438394546509, "sampling/sampling_logp_difference/max": 1.444122314453125, "sampling/importance_sampling_ratio/min": 0.23595307767391205, "sampling/importance_sampling_ratio/mean": 1.012911319732666, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7153090685606003, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09122720453888178, "clip_ratio/high_max": 0.09122720453888178, "clip_ratio/region_mean": 0.09122720453888178, "reward_total_mean": 0.6136913299560547, "reward_meter_mean": 0.8672534227371216, "reward_meter_std": 0.34703925251960754, "reward_count_adherence_mean": 0.6458333730697632, "reward_count_adherence_std": 0.20773723721504211, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8143404722213745, "reward_repeat_soft_std": 0.0835551917552948, "reward_judge_quality_mean": 0.20875000953674316, "reward_judge_quality_std": 0.11038083583116531, "reward_total_composite_mean": 0.6136913299560547, "reward_total_composite_std": 0.25135669112205505} {"timestamp_utc": "2026-04-13T00:51:44Z", "mode": "train", "global_step": 1042, "epoch": 0.10467101958814666, "loss": 0.0322, "grad_norm": 13.335729598999023, "learning_rate": 6.845454545454546e-06, "num_tokens": 1875898.0, "completions/mean_length": 43.375, "completions/min_length": 30.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.375, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.7876136302947998, "rewards/meter/std": 0.1890069544315338, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.979156494140625, "rewards/repeat_soft/std": 0.018416594713926315, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.722091794013977, "rewards/total_composite/std": 0.08130525052547455, "reward": 0.722091794013977, "reward_std": 0.08130525052547455, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.124754898250103, "sampling/sampling_logp_difference/max": 1.9666454792022705, "sampling/importance_sampling_ratio/min": 0.13992545008659363, "sampling/importance_sampling_ratio/mean": 1.0021204948425293, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7269665226340294, "clip_ratio/low_mean": 0.055597685277462006, "clip_ratio/low_min": 0.055597685277462006, "clip_ratio/high_mean": 0.05611933581531048, "clip_ratio/high_max": 0.05611933581531048, "clip_ratio/region_mean": 0.11171702109277248, "reward_total_mean": 0.722091794013977, "reward_meter_mean": 0.7876136302947998, "reward_meter_std": 0.1890069544315338, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.979156494140625, "reward_repeat_soft_std": 0.018416594713926315, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.722091794013977, "reward_total_composite_std": 0.08130525052547455} {"timestamp_utc": "2026-04-13T00:51:52Z", "mode": "train", "global_step": 1043, "epoch": 0.10477147162230036, "loss": 0.0137, "grad_norm": 8.464723587036133, "learning_rate": 6.842424242424243e-06, "num_tokens": 1878093.0, "completions/mean_length": 85.375, "completions/min_length": 82.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.375, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.8727500438690186, "rewards/meter/std": 0.14875324070453644, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9301455020904541, "rewards/repeat_soft/std": 0.05667625367641449, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.12351980060338974, "rewards/total_composite/mean": 0.6987521052360535, "rewards/total_composite/std": 0.06388669461011887, "reward": 0.6987521052360535, "reward_std": 0.06388667970895767, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1215549111366272, "sampling/sampling_logp_difference/max": 1.5919413566589355, "sampling/importance_sampling_ratio/min": 0.20353010296821594, "sampling/importance_sampling_ratio/mean": 1.019569993019104, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8405453115701675, "clip_ratio/low_mean": 0.03606415260583162, "clip_ratio/low_min": 0.03606415260583162, "clip_ratio/high_mean": 0.07058594189584255, "clip_ratio/high_max": 0.07058594189584255, "clip_ratio/region_mean": 0.10665009450167418, "reward_total_mean": 0.6987521052360535, "reward_meter_mean": 0.8727500438690186, "reward_meter_std": 0.14875324070453644, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9301455020904541, "reward_repeat_soft_std": 0.05667625367641449, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.12351980060338974, "reward_total_composite_mean": 0.6987521052360535, "reward_total_composite_std": 0.06388669461011887} {"timestamp_utc": "2026-04-13T00:52:28Z", "mode": "train", "global_step": 1044, "epoch": 0.10487192365645405, "loss": -0.0379, "grad_norm": 13.852270126342773, "learning_rate": 6.83939393939394e-06, "num_tokens": 1879878.0, "completions/mean_length": 53.125, "completions/min_length": 47.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.125, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.4434677064418793, "rewards/meter/std": 0.32767441868782043, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9418444037437439, "rewards/repeat_soft/std": 0.02954765036702156, "rewards/judge_quality/mean": 0.1587499976158142, "rewards/judge_quality/std": 0.19357076287269592, "rewards/total_composite/mean": 0.49136990308761597, "rewards/total_composite/std": 0.15760111808776855, "reward": 0.49136990308761597, "reward_std": 0.15760111808776855, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15691417455673218, "sampling/sampling_logp_difference/max": 2.899650812149048, "sampling/importance_sampling_ratio/min": 0.055042434483766556, "sampling/importance_sampling_ratio/mean": 1.024127721786499, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0573291555047035, "clip_ratio/low_mean": 0.049303000792860985, "clip_ratio/low_min": 0.049303000792860985, "clip_ratio/high_mean": 0.0631719846278429, "clip_ratio/high_max": 0.0631719846278429, "clip_ratio/region_mean": 0.11247498542070389, "reward_total_mean": 0.49136990308761597, "reward_meter_mean": 0.4434677064418793, "reward_meter_std": 0.32767441868782043, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9418444037437439, "reward_repeat_soft_std": 0.02954765036702156, "reward_judge_quality_mean": 0.1587499976158142, "reward_judge_quality_std": 0.19357076287269592, "reward_total_composite_mean": 0.49136990308761597, "reward_total_composite_std": 0.15760111808776855} {"timestamp_utc": "2026-04-13T00:52:37Z", "mode": "train", "global_step": 1045, "epoch": 0.10497237569060773, "loss": 0.107, "grad_norm": 22.523542404174805, "learning_rate": 6.8363636363636364e-06, "num_tokens": 1881461.0, "completions/mean_length": 29.875, "completions/min_length": 25.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.875, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.8736658096313477, "rewards/meter/std": 0.34033632278442383, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9468586444854736, "rewards/repeat_soft/std": 0.024185575544834137, "rewards/judge_quality/mean": 0.38374999165534973, "rewards/judge_quality/std": 0.11697831749916077, "rewards/total_composite/mean": 0.7529604434967041, "rewards/total_composite/std": 0.1485629826784134, "reward": 0.7529604434967041, "reward_std": 0.1485629826784134, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14548487961292267, "sampling/sampling_logp_difference/max": 1.305485725402832, "sampling/importance_sampling_ratio/min": 0.2710408568382263, "sampling/importance_sampling_ratio/mean": 0.9978763461112976, "sampling/importance_sampling_ratio/max": 1.7578798532485962, "entropy": 1.006789281964302, "clip_ratio/low_mean": 0.025000000838190317, "clip_ratio/low_min": 0.025000000838190317, "clip_ratio/high_mean": 0.14303261041641235, "clip_ratio/high_max": 0.14303261041641235, "clip_ratio/region_mean": 0.16803261125460267, "reward_total_mean": 0.7529604434967041, "reward_meter_mean": 0.8736658096313477, "reward_meter_std": 0.34033632278442383, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9468586444854736, "reward_repeat_soft_std": 0.024185575544834137, "reward_judge_quality_mean": 0.38374999165534973, "reward_judge_quality_std": 0.11697831749916077, "reward_total_composite_mean": 0.7529604434967041, "reward_total_composite_std": 0.1485629826784134} {"timestamp_utc": "2026-04-13T00:52:44Z", "mode": "train", "global_step": 1046, "epoch": 0.10507282772476143, "loss": 0.0473, "grad_norm": 10.326911926269531, "learning_rate": 6.833333333333334e-06, "num_tokens": 1883143.0, "completions/mean_length": 49.25, "completions/min_length": 44.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.25, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.9900477528572083, "rewards/meter/std": 0.008088136091828346, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9833897352218628, "rewards/repeat_soft/std": 0.0175783671438694, "rewards/judge_quality/mean": 0.48000001907348633, "rewards/judge_quality/std": 0.10993505269289017, "rewards/total_composite/mean": 0.7878605127334595, "rewards/total_composite/std": 0.03392486274242401, "reward": 0.7878605127334595, "reward_std": 0.03392487391829491, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1456579566001892, "sampling/sampling_logp_difference/max": 2.9719297885894775, "sampling/importance_sampling_ratio/min": 0.05120440199971199, "sampling/importance_sampling_ratio/mean": 1.0280070304870605, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9428757131099701, "clip_ratio/low_mean": 0.12396427523344755, "clip_ratio/low_min": 0.12396427523344755, "clip_ratio/high_mean": 0.013586956076323986, "clip_ratio/high_max": 0.013586956076323986, "clip_ratio/region_mean": 0.13755123130977154, "reward_total_mean": 0.7878605127334595, "reward_meter_mean": 0.9900477528572083, "reward_meter_std": 0.008088136091828346, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9833897352218628, "reward_repeat_soft_std": 0.0175783671438694, "reward_judge_quality_mean": 0.48000001907348633, "reward_judge_quality_std": 0.10993505269289017, "reward_total_composite_mean": 0.7878605127334595, "reward_total_composite_std": 0.03392486274242401} {"timestamp_utc": "2026-04-13T00:52:51Z", "mode": "train", "global_step": 1047, "epoch": 0.10517327975891512, "loss": 0.0929, "grad_norm": 21.016836166381836, "learning_rate": 6.83030303030303e-06, "num_tokens": 1884531.0, "completions/mean_length": 27.5, "completions/min_length": 21.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.5, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9426379203796387, "rewards/meter/std": 0.13063576817512512, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9503588080406189, "rewards/repeat_soft/std": 0.02873852103948593, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7974729537963867, "rewards/total_composite/std": 0.05951424315571785, "reward": 0.7974729537963867, "reward_std": 0.05951423570513725, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19725780189037323, "sampling/sampling_logp_difference/max": 2.1411361694335938, "sampling/importance_sampling_ratio/min": 0.1175212487578392, "sampling/importance_sampling_ratio/mean": 0.9987015128135681, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.034657597541809, "clip_ratio/low_mean": 0.02734375, "clip_ratio/low_min": 0.02734375, "clip_ratio/high_mean": 0.1684689363464713, "clip_ratio/high_max": 0.1684689363464713, "clip_ratio/region_mean": 0.1958126863464713, "reward_total_mean": 0.7974729537963867, "reward_meter_mean": 0.9426379203796387, "reward_meter_std": 0.13063576817512512, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9503588080406189, "reward_repeat_soft_std": 0.02873852103948593, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7974729537963867, "reward_total_composite_std": 0.05951424315571785} {"timestamp_utc": "2026-04-13T00:52:58Z", "mode": "train", "global_step": 1048, "epoch": 0.10527373179306881, "loss": 0.0483, "grad_norm": 12.890825271606445, "learning_rate": 6.827272727272728e-06, "num_tokens": 1886139.0, "completions/mean_length": 44.0, "completions/min_length": 31.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.9883289933204651, "rewards/meter/std": 0.004062440246343613, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9747357964515686, "rewards/repeat_soft/std": 0.022467322647571564, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.8079715967178345, "rewards/total_composite/std": 0.06891082972288132, "reward": 0.8079715967178345, "reward_std": 0.06891082972288132, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15011000633239746, "sampling/sampling_logp_difference/max": 1.2831683158874512, "sampling/importance_sampling_ratio/min": 0.2771577835083008, "sampling/importance_sampling_ratio/mean": 1.0185173749923706, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9217648431658745, "clip_ratio/low_mean": 0.10092940181493759, "clip_ratio/low_min": 0.10092940181493759, "clip_ratio/high_mean": 0.052489178255200386, "clip_ratio/high_max": 0.052489178255200386, "clip_ratio/region_mean": 0.15341858007013798, "reward_total_mean": 0.8079715967178345, "reward_meter_mean": 0.9883289933204651, "reward_meter_std": 0.004062440246343613, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9747357964515686, "reward_repeat_soft_std": 0.022467322647571564, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.8079715967178345, "reward_total_composite_std": 0.06891082972288132} {"timestamp_utc": "2026-04-13T00:53:05Z", "mode": "train", "global_step": 1049, "epoch": 0.1053741838272225, "loss": 0.1077, "grad_norm": 17.264419555664062, "learning_rate": 6.824242424242425e-06, "num_tokens": 1887499.0, "completions/mean_length": 20.0, "completions/min_length": 16.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.0, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.9881855845451355, "rewards/meter/std": 0.004739299416542053, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9268490076065063, "rewards/repeat_soft/std": 0.03339920938014984, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.8069933652877808, "rewards/total_composite/std": 0.02028741128742695, "reward": 0.8069933652877808, "reward_std": 0.020287418738007545, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11587798595428467, "sampling/sampling_logp_difference/max": 1.112144947052002, "sampling/importance_sampling_ratio/min": 0.4100440442562103, "sampling/importance_sampling_ratio/mean": 1.042249083518982, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8575710505247116, "clip_ratio/low_mean": 0.03125, "clip_ratio/low_min": 0.03125, "clip_ratio/high_mean": 0.11512235179543495, "clip_ratio/high_max": 0.11512235179543495, "clip_ratio/region_mean": 0.14637235179543495, "reward_total_mean": 0.8069933652877808, "reward_meter_mean": 0.9881855845451355, "reward_meter_std": 0.004739299416542053, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9268490076065063, "reward_repeat_soft_std": 0.03339920938014984, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.8069933652877808, "reward_total_composite_std": 0.02028741128742695} {"timestamp_utc": "2026-04-13T00:53:12Z", "mode": "train", "global_step": 1050, "epoch": 0.10547463586137619, "loss": 0.0505, "grad_norm": 17.70060157775879, "learning_rate": 6.821212121212122e-06, "num_tokens": 1889319.0, "completions/mean_length": 56.5, "completions/min_length": 52.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.5, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.985855758190155, "rewards/meter/std": 0.014370645396411419, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9691030979156494, "rewards/repeat_soft/std": 0.033968716859817505, "rewards/judge_quality/mean": 0.4024999737739563, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.7612953782081604, "rewards/total_composite/std": 0.023677010089159012, "reward": 0.7612953782081604, "reward_std": 0.02367701195180416, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14524701237678528, "sampling/sampling_logp_difference/max": 2.6232380867004395, "sampling/importance_sampling_ratio/min": 0.07256750017404556, "sampling/importance_sampling_ratio/mean": 1.0018717050552368, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7391493692994118, "clip_ratio/low_mean": 0.01875000074505806, "clip_ratio/low_min": 0.01875000074505806, "clip_ratio/high_mean": 0.14057897217571735, "clip_ratio/high_max": 0.14057897217571735, "clip_ratio/region_mean": 0.1593289729207754, "reward_total_mean": 0.7612953782081604, "reward_meter_mean": 0.985855758190155, "reward_meter_std": 0.014370645396411419, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9691030979156494, "reward_repeat_soft_std": 0.033968716859817505, "reward_judge_quality_mean": 0.4024999737739563, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.7612953782081604, "reward_total_composite_std": 0.023677010089159012} {"timestamp_utc": "2026-04-13T00:54:06Z", "mode": "eval", "global_step": 1050, "epoch": 0.10547463586137619, "eval_loss": NaN, "eval_runtime": 53.6279, "eval_samples_per_second": 1.492, "eval_steps_per_second": 0.186, "eval_num_tokens": 1889319.0, "eval_completions/mean_length": 68.65, "eval_completions/min_length": 33.4, "eval_completions/max_length": 151.3, "eval_completions/clipped_ratio": 0.0125, "eval_completions/mean_terminated_length": 62.97321472167969, "eval_completions/min_terminated_length": 33.4, "eval_completions/max_terminated_length": 107.8, "eval_rewards/meter/mean": 0.7794733345508575, "eval_rewards/meter/std": 0.2912589253857732, "eval_rewards/count_adherence/mean": 0.8287499964237213, "eval_rewards/count_adherence/std": 0.14322784841060637, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.07071067690849304, "eval_rewards/repeat_soft/mean": 0.9468573808670044, "eval_rewards/repeat_soft/std": 0.052804647199809554, "eval_rewards/judge_quality/mean": 0.435124996304512, "eval_rewards/judge_quality/std": 0.15623055025935173, "eval_rewards/total_composite/mean": 0.6893376469612121, "eval_rewards/total_composite/std": 0.16441902965307237, "eval_reward": 0.6893376469612121, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.07511443197727204, "eval_sampling/sampling_logp_difference/max": 0.9342038631439209, "eval_sampling/importance_sampling_ratio/min": 0.3994178682565689, "eval_sampling/importance_sampling_ratio/mean": 1.0169029712677002, "eval_sampling/importance_sampling_ratio/max": 1.4962864756584167, "eval_entropy": 0.8591082930564881, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6893376469612121, "eval_reward_meter_mean": 0.7794733345508575, "eval_reward_meter_std": 0.2912589253857732, "eval_reward_count_adherence_mean": 0.8287499964237213, "eval_reward_count_adherence_std": 0.14322784841060637, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.07071067690849304, "eval_reward_repeat_soft_mean": 0.9468573808670044, "eval_reward_repeat_soft_std": 0.052804647199809554, "eval_reward_judge_quality_mean": 0.435124996304512, "eval_reward_judge_quality_std": 0.15623055025935173, "eval_reward_total_composite_mean": 0.6893376469612121, "eval_reward_total_composite_std": 0.16441902965307237} {"timestamp_utc": "2026-04-13T00:54:16Z", "mode": "train", "global_step": 1051, "epoch": 0.10557508789552988, "loss": 0.0003, "grad_norm": 23.728622436523438, "learning_rate": 6.818181818181818e-06, "num_tokens": 1890688.0, "completions/mean_length": 22.125, "completions/min_length": 15.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.125, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9809482097625732, "rewards/meter/std": 0.011744909919798374, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9493104219436646, "rewards/repeat_soft/std": 0.011101635172963142, "rewards/judge_quality/mean": 0.5487499833106995, "rewards/judge_quality/std": 0.22937417030334473, "rewards/total_composite/mean": 0.8509827256202698, "rewards/total_composite/std": 0.06966744363307953, "reward": 0.8509827256202698, "reward_std": 0.06966746598482132, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15760402381420135, "sampling/sampling_logp_difference/max": 1.7945530414581299, "sampling/importance_sampling_ratio/min": 0.16620171070098877, "sampling/importance_sampling_ratio/mean": 0.9940311312675476, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6714931949973106, "clip_ratio/low_mean": 0.08380952617153525, "clip_ratio/low_min": 0.08380952617153525, "clip_ratio/high_mean": 0.0163043481297791, "clip_ratio/high_max": 0.0163043481297791, "clip_ratio/region_mean": 0.10011387430131435, "reward_total_mean": 0.8509827256202698, "reward_meter_mean": 0.9809482097625732, "reward_meter_std": 0.011744909919798374, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9493104219436646, "reward_repeat_soft_std": 0.011101635172963142, "reward_judge_quality_mean": 0.5487499833106995, "reward_judge_quality_std": 0.22937417030334473, "reward_total_composite_mean": 0.8509827256202698, "reward_total_composite_std": 0.06966744363307953} {"timestamp_utc": "2026-04-13T00:54:24Z", "mode": "train", "global_step": 1052, "epoch": 0.10567553992968358, "loss": 0.1018, "grad_norm": 7.012650966644287, "learning_rate": 6.8151515151515155e-06, "num_tokens": 1893032.0, "completions/mean_length": 116.0, "completions/min_length": 90.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.0, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.989003598690033, "rewards/meter/std": 0.013667004182934761, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7972731590270996, "rewards/repeat_soft/std": 0.21851561963558197, "rewards/judge_quality/mean": 0.30124998092651367, "rewards/judge_quality/std": 0.10398317128419876, "rewards/total_composite/mean": 0.7276539206504822, "rewards/total_composite/std": 0.04628615453839302, "reward": 0.7276539206504822, "reward_std": 0.04628613963723183, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11959505081176758, "sampling/sampling_logp_difference/max": 2.3658037185668945, "sampling/importance_sampling_ratio/min": 0.09387382119894028, "sampling/importance_sampling_ratio/mean": 1.005288004875183, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8640038818120956, "clip_ratio/low_mean": 0.03265109099447727, "clip_ratio/low_min": 0.03265109099447727, "clip_ratio/high_mean": 0.06458384543657303, "clip_ratio/high_max": 0.06458384543657303, "clip_ratio/region_mean": 0.0972349364310503, "reward_total_mean": 0.7276539206504822, "reward_meter_mean": 0.989003598690033, "reward_meter_std": 0.013667004182934761, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7972731590270996, "reward_repeat_soft_std": 0.21851561963558197, "reward_judge_quality_mean": 0.30124998092651367, "reward_judge_quality_std": 0.10398317128419876, "reward_total_composite_mean": 0.7276539206504822, "reward_total_composite_std": 0.04628615453839302} {"timestamp_utc": "2026-04-13T00:54:32Z", "mode": "train", "global_step": 1053, "epoch": 0.10577599196383727, "loss": -0.0394, "grad_norm": 11.435787200927734, "learning_rate": 6.812121212121212e-06, "num_tokens": 1895009.0, "completions/mean_length": 68.125, "completions/min_length": 46.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.125, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.8042130470275879, "rewards/meter/std": 0.3069063723087311, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9528941512107849, "rewards/repeat_soft/std": 0.024220237508416176, "rewards/judge_quality/mean": 0.3387500047683716, "rewards/judge_quality/std": 0.09538455307483673, "rewards/total_composite/mean": 0.671310305595398, "rewards/total_composite/std": 0.14964322745800018, "reward": 0.671310305595398, "reward_std": 0.14964322745800018, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15581491589546204, "sampling/sampling_logp_difference/max": 1.927217960357666, "sampling/importance_sampling_ratio/min": 0.14555257558822632, "sampling/importance_sampling_ratio/mean": 1.0096242427825928, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9588113203644753, "clip_ratio/low_mean": 0.02580128237605095, "clip_ratio/low_min": 0.02580128237605095, "clip_ratio/high_mean": 0.09324168413877487, "clip_ratio/high_max": 0.09324168413877487, "clip_ratio/region_mean": 0.11904296651482582, "reward_total_mean": 0.671310305595398, "reward_meter_mean": 0.8042130470275879, "reward_meter_std": 0.3069063723087311, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9528941512107849, "reward_repeat_soft_std": 0.024220237508416176, "reward_judge_quality_mean": 0.3387500047683716, "reward_judge_quality_std": 0.09538455307483673, "reward_total_composite_mean": 0.671310305595398, "reward_total_composite_std": 0.14964322745800018} {"timestamp_utc": "2026-04-13T00:54:40Z", "mode": "train", "global_step": 1054, "epoch": 0.10587644399799095, "loss": -0.0395, "grad_norm": 13.104318618774414, "learning_rate": 6.80909090909091e-06, "num_tokens": 1896724.0, "completions/mean_length": 47.375, "completions/min_length": 39.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.375, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9410066604614258, "rewards/meter/std": 0.13547497987747192, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9378410577774048, "rewards/repeat_soft/std": 0.037511877715587616, "rewards/judge_quality/mean": 0.38875001668930054, "rewards/judge_quality/std": 0.08675704896450043, "rewards/total_composite/mean": 0.7838621139526367, "rewards/total_composite/std": 0.060608070343732834, "reward": 0.7838621139526367, "reward_std": 0.06060807779431343, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1600659042596817, "sampling/sampling_logp_difference/max": 1.7379450798034668, "sampling/importance_sampling_ratio/min": 0.17588144540786743, "sampling/importance_sampling_ratio/mean": 1.0262138843536377, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0864503681659698, "clip_ratio/low_mean": 0.07192226499319077, "clip_ratio/low_min": 0.07192226499319077, "clip_ratio/high_mean": 0.09447167906910181, "clip_ratio/high_max": 0.09447167906910181, "clip_ratio/region_mean": 0.16639394406229258, "reward_total_mean": 0.7838621139526367, "reward_meter_mean": 0.9410066604614258, "reward_meter_std": 0.13547497987747192, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9378410577774048, "reward_repeat_soft_std": 0.037511877715587616, "reward_judge_quality_mean": 0.38875001668930054, "reward_judge_quality_std": 0.08675704896450043, "reward_total_composite_mean": 0.7838621139526367, "reward_total_composite_std": 0.060608070343732834} {"timestamp_utc": "2026-04-13T00:54:47Z", "mode": "train", "global_step": 1055, "epoch": 0.10597689603214465, "loss": -0.0229, "grad_norm": 22.852928161621094, "learning_rate": 6.806060606060607e-06, "num_tokens": 1898252.0, "completions/mean_length": 31.0, "completions/min_length": 21.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.0, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.5594446063041687, "rewards/meter/std": 0.41683027148246765, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9809660911560059, "rewards/repeat_soft/std": 0.03281380981206894, "rewards/judge_quality/mean": 0.4637500047683716, "rewards/judge_quality/std": 0.21084439754486084, "rewards/total_composite/mean": 0.6389716863632202, "rewards/total_composite/std": 0.17530381679534912, "reward": 0.6389716863632202, "reward_std": 0.17530380189418793, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1607924997806549, "sampling/sampling_logp_difference/max": 2.681591033935547, "sampling/importance_sampling_ratio/min": 0.06845415383577347, "sampling/importance_sampling_ratio/mean": 1.0137509107589722, "sampling/importance_sampling_ratio/max": 1.9430015087127686, "entropy": 0.8383859470486641, "clip_ratio/low_mean": 0.05448066350072622, "clip_ratio/low_min": 0.05448066350072622, "clip_ratio/high_mean": 0.06856537517160177, "clip_ratio/high_max": 0.06856537517160177, "clip_ratio/region_mean": 0.123046038672328, "reward_total_mean": 0.6389716863632202, "reward_meter_mean": 0.5594446063041687, "reward_meter_std": 0.41683027148246765, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9809660911560059, "reward_repeat_soft_std": 0.03281380981206894, "reward_judge_quality_mean": 0.4637500047683716, "reward_judge_quality_std": 0.21084439754486084, "reward_total_composite_mean": 0.6389716863632202, "reward_total_composite_std": 0.17530381679534912} {"timestamp_utc": "2026-04-13T00:55:00Z", "mode": "train", "global_step": 1056, "epoch": 0.10607734806629834, "loss": -0.0769, "grad_norm": 3.947200298309326, "learning_rate": 6.803030303030304e-06, "num_tokens": 1899831.0, "completions/mean_length": 102.375, "completions/min_length": 28.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 43.85714340209961, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.27600058913230896, "rewards/meter/std": 0.2318151742219925, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.17251639068126678, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9904391765594482, "rewards/repeat_soft/std": 0.011883550323545933, "rewards/judge_quality/mean": 0.6025000214576721, "rewards/judge_quality/std": 0.3584390878677368, "rewards/total_composite/mean": 0.5081380605697632, "rewards/total_composite/std": 0.2532515227794647, "reward": 0.5081380605697632, "reward_std": 0.2532515227794647, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19415853917598724, "sampling/sampling_logp_difference/max": 1.9959008693695068, "sampling/importance_sampling_ratio/min": 0.13589118421077728, "sampling/importance_sampling_ratio/mean": 1.0128332376480103, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0574980154633522, "clip_ratio/low_mean": 0.06382978521287441, "clip_ratio/low_min": 0.06382978521287441, "clip_ratio/high_mean": 0.11590811237692833, "clip_ratio/high_max": 0.11590811237692833, "clip_ratio/region_mean": 0.17973789758980274, "reward_total_mean": 0.5081380605697632, "reward_meter_mean": 0.27600058913230896, "reward_meter_std": 0.2318151742219925, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.17251639068126678, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9904391765594482, "reward_repeat_soft_std": 0.011883550323545933, "reward_judge_quality_mean": 0.6025000214576721, "reward_judge_quality_std": 0.3584390878677368, "reward_total_composite_mean": 0.5081380605697632, "reward_total_composite_std": 0.2532515227794647} {"timestamp_utc": "2026-04-13T00:55:08Z", "mode": "train", "global_step": 1057, "epoch": 0.10617780010045204, "loss": 0.1066, "grad_norm": 12.152074813842773, "learning_rate": 6.800000000000001e-06, "num_tokens": 1901569.0, "completions/mean_length": 54.25, "completions/min_length": 45.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.25, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.6042735576629639, "rewards/meter/std": 0.4909210205078125, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8995249271392822, "rewards/repeat_soft/std": 0.06793432682752609, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5382930040359497, "rewards/total_composite/std": 0.3058469891548157, "reward": 0.5382930040359497, "reward_std": 0.3058469891548157, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14340390264987946, "sampling/sampling_logp_difference/max": 2.9481687545776367, "sampling/importance_sampling_ratio/min": 0.05243564397096634, "sampling/importance_sampling_ratio/mean": 1.0053186416625977, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7079722434282303, "clip_ratio/low_mean": 0.04861494991928339, "clip_ratio/low_min": 0.04861494991928339, "clip_ratio/high_mean": 0.08296959102153778, "clip_ratio/high_max": 0.08296959102153778, "clip_ratio/region_mean": 0.13158454094082117, "reward_total_mean": 0.5382930040359497, "reward_meter_mean": 0.6042735576629639, "reward_meter_std": 0.4909210205078125, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8995249271392822, "reward_repeat_soft_std": 0.06793432682752609, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5382930040359497, "reward_total_composite_std": 0.3058469891548157} {"timestamp_utc": "2026-04-13T00:55:16Z", "mode": "train", "global_step": 1058, "epoch": 0.10627825213460572, "loss": 0.0979, "grad_norm": 10.564192771911621, "learning_rate": 6.796969696969697e-06, "num_tokens": 1904296.0, "completions/mean_length": 113.875, "completions/min_length": 79.0, "completions/max_length": 157.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.875, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.81416916847229, "rewards/meter/std": 0.33131495118141174, "rewards/count_adherence/mean": 0.7291666865348816, "rewards/count_adherence/std": 0.0862581729888916, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9008584022521973, "rewards/repeat_soft/std": 0.05474758893251419, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.6854619979858398, "rewards/total_composite/std": 0.14433102309703827, "reward": 0.6854619979858398, "reward_std": 0.14433100819587708, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11852812767028809, "sampling/sampling_logp_difference/max": 3.308062791824341, "sampling/importance_sampling_ratio/min": 0.036586981266736984, "sampling/importance_sampling_ratio/mean": 0.9999894499778748, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6538885049521923, "clip_ratio/low_mean": 0.025960061699151993, "clip_ratio/low_min": 0.025960061699151993, "clip_ratio/high_mean": 0.0920501141808927, "clip_ratio/high_max": 0.0920501141808927, "clip_ratio/region_mean": 0.1180101758800447, "reward_total_mean": 0.6854619979858398, "reward_meter_mean": 0.81416916847229, "reward_meter_std": 0.33131495118141174, "reward_count_adherence_mean": 0.7291666865348816, "reward_count_adherence_std": 0.0862581729888916, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9008584022521973, "reward_repeat_soft_std": 0.05474758893251419, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.6854619979858398, "reward_total_composite_std": 0.14433102309703827} {"timestamp_utc": "2026-04-13T00:55:23Z", "mode": "train", "global_step": 1059, "epoch": 0.10637870416875941, "loss": 0.0147, "grad_norm": 13.836250305175781, "learning_rate": 6.793939393939395e-06, "num_tokens": 1905808.0, "completions/mean_length": 32.0, "completions/min_length": 28.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9921358227729797, "rewards/meter/std": 0.0021170848049223423, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9556125402450562, "rewards/repeat_soft/std": 0.019480695948004723, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.1940544992685318, "rewards/total_composite/mean": 0.8315223455429077, "rewards/total_composite/std": 0.05838419124484062, "reward": 0.8315223455429077, "reward_std": 0.05838420242071152, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12782928347587585, "sampling/sampling_logp_difference/max": 1.3578591346740723, "sampling/importance_sampling_ratio/min": 0.2572108209133148, "sampling/importance_sampling_ratio/mean": 1.0138490200042725, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7816659212112427, "clip_ratio/low_mean": 0.11739878100343049, "clip_ratio/low_min": 0.11739878100343049, "clip_ratio/high_mean": 0.014285714365541935, "clip_ratio/high_max": 0.014285714365541935, "clip_ratio/region_mean": 0.13168449536897242, "reward_total_mean": 0.8315223455429077, "reward_meter_mean": 0.9921358227729797, "reward_meter_std": 0.0021170848049223423, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9556125402450562, "reward_repeat_soft_std": 0.019480695948004723, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.1940544992685318, "reward_total_composite_mean": 0.8315223455429077, "reward_total_composite_std": 0.05838419124484062} {"timestamp_utc": "2026-04-13T00:55:31Z", "mode": "train", "global_step": 1060, "epoch": 0.10647915620291311, "loss": -0.0058, "grad_norm": 11.673070907592773, "learning_rate": 6.790909090909091e-06, "num_tokens": 1907750.0, "completions/mean_length": 62.75, "completions/min_length": 44.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.75, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9890538454055786, "rewards/meter/std": 0.005095128435641527, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9422520995140076, "rewards/repeat_soft/std": 0.044338349252939224, "rewards/judge_quality/mean": 0.5137500166893005, "rewards/judge_quality/std": 0.1277204155921936, "rewards/total_composite/mean": 0.8059244155883789, "rewards/total_composite/std": 0.04034656658768654, "reward": 0.8059244155883789, "reward_std": 0.040346577763557434, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13995201885700226, "sampling/sampling_logp_difference/max": 1.4963483810424805, "sampling/importance_sampling_ratio/min": 0.2239464521408081, "sampling/importance_sampling_ratio/mean": 1.0239185094833374, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9119938686490059, "clip_ratio/low_mean": 0.10303323529660702, "clip_ratio/low_min": 0.10303323529660702, "clip_ratio/high_mean": 0.04253472201526165, "clip_ratio/high_max": 0.04253472201526165, "clip_ratio/region_mean": 0.14556795731186867, "reward_total_mean": 0.8059244155883789, "reward_meter_mean": 0.9890538454055786, "reward_meter_std": 0.005095128435641527, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9422520995140076, "reward_repeat_soft_std": 0.044338349252939224, "reward_judge_quality_mean": 0.5137500166893005, "reward_judge_quality_std": 0.1277204155921936, "reward_total_composite_mean": 0.8059244155883789, "reward_total_composite_std": 0.04034656658768654} {"timestamp_utc": "2026-04-13T00:55:39Z", "mode": "train", "global_step": 1061, "epoch": 0.1065796082370668, "loss": 0.1219, "grad_norm": 20.08896827697754, "learning_rate": 6.787878787878789e-06, "num_tokens": 1909356.0, "completions/mean_length": 51.75, "completions/min_length": 43.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.75, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.7067584991455078, "rewards/meter/std": 0.4202998876571655, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9761940240859985, "rewards/repeat_soft/std": 0.031179353594779968, "rewards/judge_quality/mean": 0.5349999666213989, "rewards/judge_quality/std": 0.24663449823856354, "rewards/total_composite/mean": 0.7261607050895691, "rewards/total_composite/std": 0.16318675875663757, "reward": 0.7261607050895691, "reward_std": 0.16318677365779877, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14709247648715973, "sampling/sampling_logp_difference/max": 2.1316847801208496, "sampling/importance_sampling_ratio/min": 0.11863724142313004, "sampling/importance_sampling_ratio/mean": 1.0087025165557861, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8907738775014877, "clip_ratio/low_mean": 0.027069627307355404, "clip_ratio/low_min": 0.027069627307355404, "clip_ratio/high_mean": 0.09228017367422581, "clip_ratio/high_max": 0.09228017367422581, "clip_ratio/region_mean": 0.11934980098158121, "reward_total_mean": 0.7261607050895691, "reward_meter_mean": 0.7067584991455078, "reward_meter_std": 0.4202998876571655, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9761940240859985, "reward_repeat_soft_std": 0.031179353594779968, "reward_judge_quality_mean": 0.5349999666213989, "reward_judge_quality_std": 0.24663449823856354, "reward_total_composite_mean": 0.7261607050895691, "reward_total_composite_std": 0.16318675875663757} {"timestamp_utc": "2026-04-13T00:55:47Z", "mode": "train", "global_step": 1062, "epoch": 0.1066800602712205, "loss": 0.1003, "grad_norm": 14.36760425567627, "learning_rate": 6.7848484848484855e-06, "num_tokens": 1911216.0, "completions/mean_length": 57.5, "completions/min_length": 47.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.5, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9732199907302856, "rewards/meter/std": 0.01778586581349373, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9566714763641357, "rewards/repeat_soft/std": 0.03358738496899605, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7721161842346191, "rewards/total_composite/std": 0.00804363563656807, "reward": 0.7721161842346191, "reward_std": 0.00804363377392292, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13513335585594177, "sampling/sampling_logp_difference/max": 1.2598822116851807, "sampling/importance_sampling_ratio/min": 0.28368744254112244, "sampling/importance_sampling_ratio/mean": 1.0426719188690186, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9857887774705887, "clip_ratio/low_mean": 0.047330106142908335, "clip_ratio/low_min": 0.047330106142908335, "clip_ratio/high_mean": 0.050223374273627996, "clip_ratio/high_max": 0.050223374273627996, "clip_ratio/region_mean": 0.09755348041653633, "reward_total_mean": 0.7721161842346191, "reward_meter_mean": 0.9732199907302856, "reward_meter_std": 0.01778586581349373, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9566714763641357, "reward_repeat_soft_std": 0.03358738496899605, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7721161842346191, "reward_total_composite_std": 0.00804363563656807} {"timestamp_utc": "2026-04-13T00:55:55Z", "mode": "train", "global_step": 1063, "epoch": 0.10678051230537418, "loss": -0.0338, "grad_norm": 9.187300682067871, "learning_rate": 6.781818181818183e-06, "num_tokens": 1913164.0, "completions/mean_length": 62.5, "completions/min_length": 56.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.5, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.8881810903549194, "rewards/meter/std": 0.2742149531841278, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9736993312835693, "rewards/repeat_soft/std": 0.01680658385157585, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7228013873100281, "rewards/total_composite/std": 0.1408482939004898, "reward": 0.7228013873100281, "reward_std": 0.1408482939004898, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14612889289855957, "sampling/sampling_logp_difference/max": 1.1752820014953613, "sampling/importance_sampling_ratio/min": 0.30873191356658936, "sampling/importance_sampling_ratio/mean": 1.0111005306243896, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1180159971117973, "clip_ratio/low_mean": 0.0223214291036129, "clip_ratio/low_min": 0.0223214291036129, "clip_ratio/high_mean": 0.1390654845163226, "clip_ratio/high_max": 0.1390654845163226, "clip_ratio/region_mean": 0.1613869136199355, "reward_total_mean": 0.7228013873100281, "reward_meter_mean": 0.8881810903549194, "reward_meter_std": 0.2742149531841278, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9736993312835693, "reward_repeat_soft_std": 0.01680658385157585, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7228013873100281, "reward_total_composite_std": 0.1408482939004898} {"timestamp_utc": "2026-04-13T00:56:03Z", "mode": "train", "global_step": 1064, "epoch": 0.10688096433952787, "loss": -0.0059, "grad_norm": 24.98784637451172, "learning_rate": 6.778787878787879e-06, "num_tokens": 1914727.0, "completions/mean_length": 26.375, "completions/min_length": 21.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.375, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.8612536191940308, "rewards/meter/std": 0.31842002272605896, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.7400000095367432, "rewards/judge_quality/std": 0.24859607219696045, "rewards/total_composite/mean": 0.8558141589164734, "rewards/total_composite/std": 0.19290496408939362, "reward": 0.8558141589164734, "reward_std": 0.19290496408939362, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15831109881401062, "sampling/sampling_logp_difference/max": 1.9278926849365234, "sampling/importance_sampling_ratio/min": 0.14545439183712006, "sampling/importance_sampling_ratio/mean": 1.0085904598236084, "sampling/importance_sampling_ratio/max": 1.7250487804412842, "entropy": 0.9962994679808617, "clip_ratio/low_mean": 0.06005290988832712, "clip_ratio/low_min": 0.06005290988832712, "clip_ratio/high_mean": 0.09816628228873014, "clip_ratio/high_max": 0.09816628228873014, "clip_ratio/region_mean": 0.15821919217705727, "reward_total_mean": 0.8558141589164734, "reward_meter_mean": 0.8612536191940308, "reward_meter_std": 0.31842002272605896, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.7400000095367432, "reward_judge_quality_std": 0.24859607219696045, "reward_total_composite_mean": 0.8558141589164734, "reward_total_composite_std": 0.19290496408939362} {"timestamp_utc": "2026-04-13T00:56:10Z", "mode": "train", "global_step": 1065, "epoch": 0.10698141637368157, "loss": 0.0045, "grad_norm": 9.637001037597656, "learning_rate": 6.7757575757575765e-06, "num_tokens": 1916565.0, "completions/mean_length": 62.75, "completions/min_length": 56.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.75, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9947084188461304, "rewards/meter/std": 0.0037521524354815483, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.961662769317627, "rewards/repeat_soft/std": 0.04130396619439125, "rewards/judge_quality/mean": 0.41874998807907104, "rewards/judge_quality/std": 0.21931305527687073, "rewards/total_composite/mean": 0.8194100260734558, "rewards/total_composite/std": 0.06828704476356506, "reward": 0.8194100260734558, "reward_std": 0.06828706711530685, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12420839071273804, "sampling/sampling_logp_difference/max": 1.4644255638122559, "sampling/importance_sampling_ratio/min": 0.23121076822280884, "sampling/importance_sampling_ratio/mean": 1.0194145441055298, "sampling/importance_sampling_ratio/max": 1.9690073728561401, "entropy": 0.947543129324913, "clip_ratio/low_mean": 0.0678348233923316, "clip_ratio/low_min": 0.0678348233923316, "clip_ratio/high_mean": 0.04697942826896906, "clip_ratio/high_max": 0.04697942826896906, "clip_ratio/region_mean": 0.11481425166130066, "reward_total_mean": 0.8194100260734558, "reward_meter_mean": 0.9947084188461304, "reward_meter_std": 0.0037521524354815483, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.961662769317627, "reward_repeat_soft_std": 0.04130396619439125, "reward_judge_quality_mean": 0.41874998807907104, "reward_judge_quality_std": 0.21931305527687073, "reward_total_composite_mean": 0.8194100260734558, "reward_total_composite_std": 0.06828704476356506} {"timestamp_utc": "2026-04-13T00:56:18Z", "mode": "train", "global_step": 1066, "epoch": 0.10708186840783526, "loss": 0.0219, "grad_norm": 9.242881774902344, "learning_rate": 6.772727272727273e-06, "num_tokens": 1918766.0, "completions/mean_length": 82.125, "completions/min_length": 63.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.125, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.8357464075088501, "rewards/meter/std": 0.2965885102748871, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9534140825271606, "rewards/repeat_soft/std": 0.04437160864472389, "rewards/judge_quality/mean": 0.44749999046325684, "rewards/judge_quality/std": 0.12848013639450073, "rewards/total_composite/mean": 0.7181772589683533, "rewards/total_composite/std": 0.10129229724407196, "reward": 0.7181772589683533, "reward_std": 0.10129231214523315, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16295002400875092, "sampling/sampling_logp_difference/max": 10.021646499633789, "sampling/importance_sampling_ratio/min": 4.442774297785945e-05, "sampling/importance_sampling_ratio/mean": 1.0105260610580444, "sampling/importance_sampling_ratio/max": 1.9561951160430908, "entropy": 0.8551634922623634, "clip_ratio/low_mean": 0.052489178255200386, "clip_ratio/low_min": 0.052489178255200386, "clip_ratio/high_mean": 0.06914125103503466, "clip_ratio/high_max": 0.06914125103503466, "clip_ratio/region_mean": 0.12163042929023504, "reward_total_mean": 0.7181772589683533, "reward_meter_mean": 0.8357464075088501, "reward_meter_std": 0.2965885102748871, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9534140825271606, "reward_repeat_soft_std": 0.04437160864472389, "reward_judge_quality_mean": 0.44749999046325684, "reward_judge_quality_std": 0.12848013639450073, "reward_total_composite_mean": 0.7181772589683533, "reward_total_composite_std": 0.10129229724407196} {"timestamp_utc": "2026-04-13T00:56:24Z", "mode": "train", "global_step": 1067, "epoch": 0.10718232044198896, "loss": 0.0989, "grad_norm": 25.568086624145508, "learning_rate": 6.76969696969697e-06, "num_tokens": 1920300.0, "completions/mean_length": 32.75, "completions/min_length": 28.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.75, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.753685474395752, "rewards/meter/std": 0.3112368881702423, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9801715016365051, "rewards/repeat_soft/std": 0.02842319943010807, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7165506482124329, "rewards/total_composite/std": 0.13514752686023712, "reward": 0.7165506482124329, "reward_std": 0.13514754176139832, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20423489809036255, "sampling/sampling_logp_difference/max": 1.6089587211608887, "sampling/importance_sampling_ratio/min": 0.20009586215019226, "sampling/importance_sampling_ratio/mean": 1.0283622741699219, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9304540231823921, "clip_ratio/low_mean": 0.0689384862780571, "clip_ratio/low_min": 0.0689384862780571, "clip_ratio/high_mean": 0.0976550942286849, "clip_ratio/high_max": 0.0976550942286849, "clip_ratio/region_mean": 0.166593580506742, "reward_total_mean": 0.7165506482124329, "reward_meter_mean": 0.753685474395752, "reward_meter_std": 0.3112368881702423, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9801715016365051, "reward_repeat_soft_std": 0.02842319943010807, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7165506482124329, "reward_total_composite_std": 0.13514752686023712} {"timestamp_utc": "2026-04-13T00:56:31Z", "mode": "train", "global_step": 1068, "epoch": 0.10728277247614264, "loss": 0.0274, "grad_norm": 11.295289993286133, "learning_rate": 6.7666666666666665e-06, "num_tokens": 1922111.0, "completions/mean_length": 66.375, "completions/min_length": 63.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.375, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9559487104415894, "rewards/meter/std": 0.037048399448394775, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9759101271629333, "rewards/repeat_soft/std": 0.020949002355337143, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.7850179672241211, "rewards/total_composite/std": 0.04975026100873947, "reward": 0.7850179672241211, "reward_std": 0.04975024610757828, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1495111733675003, "sampling/sampling_logp_difference/max": 1.4928646087646484, "sampling/importance_sampling_ratio/min": 0.22472797334194183, "sampling/importance_sampling_ratio/mean": 1.008102297782898, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8830416798591614, "clip_ratio/low_mean": 0.11267613153904676, "clip_ratio/low_min": 0.11267613153904676, "clip_ratio/high_mean": 0.03341427445411682, "clip_ratio/high_max": 0.03341427445411682, "clip_ratio/region_mean": 0.14609040599316359, "reward_total_mean": 0.7850179672241211, "reward_meter_mean": 0.9559487104415894, "reward_meter_std": 0.037048399448394775, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9759101271629333, "reward_repeat_soft_std": 0.020949002355337143, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.7850179672241211, "reward_total_composite_std": 0.04975026100873947} {"timestamp_utc": "2026-04-13T00:56:39Z", "mode": "train", "global_step": 1069, "epoch": 0.10738322451029633, "loss": 0.0697, "grad_norm": 20.45298957824707, "learning_rate": 6.763636363636365e-06, "num_tokens": 1923734.0, "completions/mean_length": 47.875, "completions/min_length": 44.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.875, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.7173110246658325, "rewards/meter/std": 0.3528819680213928, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9884952306747437, "rewards/repeat_soft/std": 0.014666780829429626, "rewards/judge_quality/mean": 0.5774999856948853, "rewards/judge_quality/std": 0.22461079061031342, "rewards/total_composite/mean": 0.744889497756958, "rewards/total_composite/std": 0.148996040225029, "reward": 0.744889497756958, "reward_std": 0.148996040225029, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15242305397987366, "sampling/sampling_logp_difference/max": 3.0052809715270996, "sampling/importance_sampling_ratio/min": 0.04952483996748924, "sampling/importance_sampling_ratio/mean": 0.9944461584091187, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6561531201004982, "clip_ratio/low_mean": 0.05374999903142452, "clip_ratio/low_min": 0.05374999903142452, "clip_ratio/high_mean": 0.07711645308881998, "clip_ratio/high_max": 0.07711645308881998, "clip_ratio/region_mean": 0.1308664521202445, "reward_total_mean": 0.744889497756958, "reward_meter_mean": 0.7173110246658325, "reward_meter_std": 0.3528819680213928, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9884952306747437, "reward_repeat_soft_std": 0.014666780829429626, "reward_judge_quality_mean": 0.5774999856948853, "reward_judge_quality_std": 0.22461079061031342, "reward_total_composite_mean": 0.744889497756958, "reward_total_composite_std": 0.148996040225029} {"timestamp_utc": "2026-04-13T00:56:46Z", "mode": "train", "global_step": 1070, "epoch": 0.10748367654445003, "loss": 0.0296, "grad_norm": 11.947762489318848, "learning_rate": 6.760606060606061e-06, "num_tokens": 1925343.0, "completions/mean_length": 44.125, "completions/min_length": 41.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.125, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9860657453536987, "rewards/meter/std": 0.012630265206098557, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9438905715942383, "rewards/repeat_soft/std": 0.02489352785050869, "rewards/judge_quality/mean": 0.7699999809265137, "rewards/judge_quality/std": 0.22677870094776154, "rewards/total_composite/mean": 0.9191185832023621, "rewards/total_composite/std": 0.06652558594942093, "reward": 0.9191185832023621, "reward_std": 0.06652558594942093, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09330005943775177, "sampling/sampling_logp_difference/max": 1.6923270225524902, "sampling/importance_sampling_ratio/min": 0.18409064412117004, "sampling/importance_sampling_ratio/mean": 0.997995138168335, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47531401365995407, "clip_ratio/low_mean": 0.02414772752672434, "clip_ratio/low_min": 0.02414772752672434, "clip_ratio/high_mean": 0.03425735607743263, "clip_ratio/high_max": 0.03425735607743263, "clip_ratio/region_mean": 0.05840508360415697, "reward_total_mean": 0.9191185832023621, "reward_meter_mean": 0.9860657453536987, "reward_meter_std": 0.012630265206098557, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9438905715942383, "reward_repeat_soft_std": 0.02489352785050869, "reward_judge_quality_mean": 0.7699999809265137, "reward_judge_quality_std": 0.22677870094776154, "reward_total_composite_mean": 0.9191185832023621, "reward_total_composite_std": 0.06652558594942093} {"timestamp_utc": "2026-04-13T00:56:53Z", "mode": "train", "global_step": 1071, "epoch": 0.10758412857860372, "loss": 0.0977, "grad_norm": 12.861830711364746, "learning_rate": 6.757575757575758e-06, "num_tokens": 1927420.0, "completions/mean_length": 63.625, "completions/min_length": 49.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.625, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.6258131265640259, "rewards/meter/std": 0.26594415307044983, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9652621150016785, "rewards/repeat_soft/std": 0.01596822589635849, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.6166421175003052, "rewards/total_composite/std": 0.11891424655914307, "reward": 0.6166421175003052, "reward_std": 0.11891423910856247, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1690264493227005, "sampling/sampling_logp_difference/max": 4.094034194946289, "sampling/importance_sampling_ratio/min": 0.016671840101480484, "sampling/importance_sampling_ratio/mean": 1.0223338603973389, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0305352285504341, "clip_ratio/low_mean": 0.06211612094193697, "clip_ratio/low_min": 0.06211612094193697, "clip_ratio/high_mean": 0.09505987912416458, "clip_ratio/high_max": 0.09505987912416458, "clip_ratio/region_mean": 0.15717600006610155, "reward_total_mean": 0.6166421175003052, "reward_meter_mean": 0.6258131265640259, "reward_meter_std": 0.26594415307044983, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9652621150016785, "reward_repeat_soft_std": 0.01596822589635849, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.6166421175003052, "reward_total_composite_std": 0.11891424655914307} {"timestamp_utc": "2026-04-13T00:57:01Z", "mode": "train", "global_step": 1072, "epoch": 0.1076845806127574, "loss": 0.0188, "grad_norm": 12.49705982208252, "learning_rate": 6.754545454545455e-06, "num_tokens": 1928990.0, "completions/mean_length": 52.25, "completions/min_length": 40.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.25, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9045677185058594, "rewards/meter/std": 0.21653713285923004, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9821046590805054, "rewards/repeat_soft/std": 0.01310367789119482, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7823910117149353, "rewards/total_composite/std": 0.09758703410625458, "reward": 0.7823910117149353, "reward_std": 0.09758703410625458, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15923188626766205, "sampling/sampling_logp_difference/max": 1.429396152496338, "sampling/importance_sampling_ratio/min": 0.2394534796476364, "sampling/importance_sampling_ratio/mean": 0.9980874061584473, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.871427372097969, "clip_ratio/low_mean": 0.019999999552965164, "clip_ratio/low_min": 0.019999999552965164, "clip_ratio/high_mean": 0.13562148995697498, "clip_ratio/high_max": 0.13562148995697498, "clip_ratio/region_mean": 0.15562148950994015, "reward_total_mean": 0.7823910117149353, "reward_meter_mean": 0.9045677185058594, "reward_meter_std": 0.21653713285923004, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9821046590805054, "reward_repeat_soft_std": 0.01310367789119482, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7823910117149353, "reward_total_composite_std": 0.09758703410625458} {"timestamp_utc": "2026-04-13T00:57:08Z", "mode": "train", "global_step": 1073, "epoch": 0.1077850326469111, "loss": 0.0255, "grad_norm": 15.451290130615234, "learning_rate": 6.751515151515152e-06, "num_tokens": 1930470.0, "completions/mean_length": 39.0, "completions/min_length": 35.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.9155656695365906, "rewards/meter/std": 0.15871933102607727, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9859304428100586, "rewards/repeat_soft/std": 0.02068207785487175, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.782472550868988, "rewards/total_composite/std": 0.07478274405002594, "reward": 0.782472550868988, "reward_std": 0.07478274405002594, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1667119860649109, "sampling/sampling_logp_difference/max": 2.039062023162842, "sampling/importance_sampling_ratio/min": 0.13015073537826538, "sampling/importance_sampling_ratio/mean": 1.0206842422485352, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2803795635700226, "clip_ratio/low_mean": 0.037454213947057724, "clip_ratio/low_min": 0.037454213947057724, "clip_ratio/high_mean": 0.1133541613817215, "clip_ratio/high_max": 0.1133541613817215, "clip_ratio/region_mean": 0.15080837532877922, "reward_total_mean": 0.782472550868988, "reward_meter_mean": 0.9155656695365906, "reward_meter_std": 0.15871933102607727, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9859304428100586, "reward_repeat_soft_std": 0.02068207785487175, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.782472550868988, "reward_total_composite_std": 0.07478274405002594} {"timestamp_utc": "2026-04-13T00:57:16Z", "mode": "train", "global_step": 1074, "epoch": 0.10788548468106479, "loss": 0.0383, "grad_norm": 17.568538665771484, "learning_rate": 6.748484848484848e-06, "num_tokens": 1932019.0, "completions/mean_length": 33.625, "completions/min_length": 28.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.625, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.6120611429214478, "rewards/meter/std": 0.39509713649749756, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.11785111576318741, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9201917052268982, "rewards/repeat_soft/std": 0.17254036664962769, "rewards/judge_quality/mean": 0.45249998569488525, "rewards/judge_quality/std": 0.21224987506866455, "rewards/total_composite/mean": 0.5788216590881348, "rewards/total_composite/std": 0.28090208768844604, "reward": 0.5788216590881348, "reward_std": 0.28090208768844604, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15735112130641937, "sampling/sampling_logp_difference/max": 1.7280073165893555, "sampling/importance_sampling_ratio/min": 0.17763803899288177, "sampling/importance_sampling_ratio/mean": 1.0130999088287354, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.071249470114708, "clip_ratio/low_mean": 0.0594470058567822, "clip_ratio/low_min": 0.0594470058567822, "clip_ratio/high_mean": 0.08960071764886379, "clip_ratio/high_max": 0.08960071764886379, "clip_ratio/region_mean": 0.149047723505646, "reward_total_mean": 0.5788216590881348, "reward_meter_mean": 0.6120611429214478, "reward_meter_std": 0.39509713649749756, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.11785111576318741, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9201917052268982, "reward_repeat_soft_std": 0.17254036664962769, "reward_judge_quality_mean": 0.45249998569488525, "reward_judge_quality_std": 0.21224987506866455, "reward_total_composite_mean": 0.5788216590881348, "reward_total_composite_std": 0.28090208768844604} {"timestamp_utc": "2026-04-13T00:57:24Z", "mode": "train", "global_step": 1075, "epoch": 0.10798593671521849, "loss": 0.078, "grad_norm": 26.53045082092285, "learning_rate": 6.7454545454545465e-06, "num_tokens": 1933650.0, "completions/mean_length": 26.875, "completions/min_length": 19.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.875, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.8683773279190063, "rewards/meter/std": 0.34260618686676025, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.32625001668930054, "rewards/judge_quality/std": 0.11350738257169724, "rewards/total_composite/mean": 0.7348948121070862, "rewards/total_composite/std": 0.17753258347511292, "reward": 0.7348948121070862, "reward_std": 0.17753256857395172, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15705358982086182, "sampling/sampling_logp_difference/max": 1.1615791320800781, "sampling/importance_sampling_ratio/min": 0.31299152970314026, "sampling/importance_sampling_ratio/mean": 1.022998332977295, "sampling/importance_sampling_ratio/max": 1.9123249053955078, "entropy": 0.9942101016640663, "clip_ratio/low_mean": 0.012500000186264515, "clip_ratio/low_min": 0.012500000186264515, "clip_ratio/high_mean": 0.13043760787695646, "clip_ratio/high_max": 0.13043760787695646, "clip_ratio/region_mean": 0.14293760806322098, "reward_total_mean": 0.7348948121070862, "reward_meter_mean": 0.8683773279190063, "reward_meter_std": 0.34260618686676025, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.32625001668930054, "reward_judge_quality_std": 0.11350738257169724, "reward_total_composite_mean": 0.7348948121070862, "reward_total_composite_std": 0.17753258347511292} {"timestamp_utc": "2026-04-13T00:57:31Z", "mode": "train", "global_step": 1076, "epoch": 0.10808638874937218, "loss": 0.1114, "grad_norm": 11.237691879272461, "learning_rate": 6.742424242424243e-06, "num_tokens": 1935315.0, "completions/mean_length": 56.125, "completions/min_length": 43.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.125, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9807167053222656, "rewards/meter/std": 0.03914443776011467, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9777017831802368, "rewards/repeat_soft/std": 0.03205263987183571, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.8360927104949951, "rewards/total_composite/std": 0.05613190680742264, "reward": 0.8360927104949951, "reward_std": 0.05613190308213234, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15027090907096863, "sampling/sampling_logp_difference/max": 2.1993589401245117, "sampling/importance_sampling_ratio/min": 0.11087421327829361, "sampling/importance_sampling_ratio/mean": 0.9846413731575012, "sampling/importance_sampling_ratio/max": 1.7739815711975098, "entropy": 0.9013178274035454, "clip_ratio/low_mean": 0.11235087085515261, "clip_ratio/low_min": 0.11235087085515261, "clip_ratio/high_mean": 0.023255813866853714, "clip_ratio/high_max": 0.023255813866853714, "clip_ratio/region_mean": 0.13560668472200632, "reward_total_mean": 0.8360927104949951, "reward_meter_mean": 0.9807167053222656, "reward_meter_std": 0.03914443776011467, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9777017831802368, "reward_repeat_soft_std": 0.03205263987183571, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.8360927104949951, "reward_total_composite_std": 0.05613190680742264} {"timestamp_utc": "2026-04-13T00:57:39Z", "mode": "train", "global_step": 1077, "epoch": 0.10818684078352586, "loss": 0.0601, "grad_norm": 11.27232837677002, "learning_rate": 6.73939393939394e-06, "num_tokens": 1937016.0, "completions/mean_length": 52.625, "completions/min_length": 42.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.625, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9558565616607666, "rewards/meter/std": 0.09808540344238281, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9378065466880798, "rewards/repeat_soft/std": 0.08714766055345535, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.8197910785675049, "rewards/total_composite/std": 0.07640678435564041, "reward": 0.8197910785675049, "reward_std": 0.07640677690505981, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14290224015712738, "sampling/sampling_logp_difference/max": 1.2428133487701416, "sampling/importance_sampling_ratio/min": 0.28857123851776123, "sampling/importance_sampling_ratio/mean": 1.0449974536895752, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3792509734630585, "clip_ratio/low_mean": 0.05998890567570925, "clip_ratio/low_min": 0.05998890567570925, "clip_ratio/high_mean": 0.0684269443154335, "clip_ratio/high_max": 0.0684269443154335, "clip_ratio/region_mean": 0.12841584999114275, "reward_total_mean": 0.8197910785675049, "reward_meter_mean": 0.9558565616607666, "reward_meter_std": 0.09808540344238281, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9378065466880798, "reward_repeat_soft_std": 0.08714766055345535, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.8197910785675049, "reward_total_composite_std": 0.07640678435564041} {"timestamp_utc": "2026-04-13T00:57:46Z", "mode": "train", "global_step": 1078, "epoch": 0.10828729281767956, "loss": 0.0668, "grad_norm": 15.460073471069336, "learning_rate": 6.7363636363636365e-06, "num_tokens": 1938815.0, "completions/mean_length": 42.875, "completions/min_length": 38.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.875, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9687113761901855, "rewards/meter/std": 0.04675973579287529, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8804298639297485, "rewards/repeat_soft/std": 0.11729412525892258, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.14520922303199768, "rewards/total_composite/mean": 0.710200309753418, "rewards/total_composite/std": 0.2906067669391632, "reward": 0.710200309753418, "reward_std": 0.2906067669391632, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15233980119228363, "sampling/sampling_logp_difference/max": 1.6121768951416016, "sampling/importance_sampling_ratio/min": 0.1994529515504837, "sampling/importance_sampling_ratio/mean": 0.9955770969390869, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.035637341439724, "clip_ratio/low_mean": 0.01822916604578495, "clip_ratio/low_min": 0.01822916604578495, "clip_ratio/high_mean": 0.14844126161187887, "clip_ratio/high_max": 0.14844126161187887, "clip_ratio/region_mean": 0.16667042765766382, "reward_total_mean": 0.710200309753418, "reward_meter_mean": 0.9687113761901855, "reward_meter_std": 0.04675973579287529, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8804298639297485, "reward_repeat_soft_std": 0.11729412525892258, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.14520922303199768, "reward_total_composite_mean": 0.710200309753418, "reward_total_composite_std": 0.2906067669391632} {"timestamp_utc": "2026-04-13T00:57:58Z", "mode": "train", "global_step": 1079, "epoch": 0.10838774485183325, "loss": -0.0339, "grad_norm": 12.301433563232422, "learning_rate": 6.733333333333334e-06, "num_tokens": 1940823.0, "completions/mean_length": 70.0, "completions/min_length": 47.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.0, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.8087680339813232, "rewards/meter/std": 0.3230033814907074, "rewards/count_adherence/mean": 0.7749999761581421, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8662908673286438, "rewards/repeat_soft/std": 0.10877832770347595, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.141421377658844, "rewards/total_composite/mean": 0.7228246927261353, "rewards/total_composite/std": 0.14231234788894653, "reward": 0.7228246927261353, "reward_std": 0.14231234788894653, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13999462127685547, "sampling/sampling_logp_difference/max": 1.9775772094726562, "sampling/importance_sampling_ratio/min": 0.13840414583683014, "sampling/importance_sampling_ratio/mean": 1.0225739479064941, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8678170815110207, "clip_ratio/low_mean": 0.03910559415817261, "clip_ratio/low_min": 0.03910559415817261, "clip_ratio/high_mean": 0.0801730495877564, "clip_ratio/high_max": 0.0801730495877564, "clip_ratio/region_mean": 0.119278643745929, "reward_total_mean": 0.7228246927261353, "reward_meter_mean": 0.8087680339813232, "reward_meter_std": 0.3230033814907074, "reward_count_adherence_mean": 0.7749999761581421, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8662908673286438, "reward_repeat_soft_std": 0.10877832770347595, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.141421377658844, "reward_total_composite_mean": 0.7228246927261353, "reward_total_composite_std": 0.14231234788894653} {"timestamp_utc": "2026-04-13T00:58:05Z", "mode": "train", "global_step": 1080, "epoch": 0.10848819688598695, "loss": -0.0001, "grad_norm": 13.783323287963867, "learning_rate": 6.73030303030303e-06, "num_tokens": 1942264.0, "completions/mean_length": 36.125, "completions/min_length": 32.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.125, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.695663571357727, "rewards/meter/std": 0.2505611181259155, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9904569983482361, "rewards/repeat_soft/std": 0.01703684590756893, "rewards/judge_quality/mean": 0.7362500429153442, "rewards/judge_quality/std": 0.25376805663108826, "rewards/total_composite/mean": 0.7829693555831909, "rewards/total_composite/std": 0.08598513901233673, "reward": 0.7829693555831909, "reward_std": 0.08598514646291733, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16519393026828766, "sampling/sampling_logp_difference/max": 2.319547653198242, "sampling/importance_sampling_ratio/min": 0.09831805527210236, "sampling/importance_sampling_ratio/mean": 1.0083650350570679, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9023748263716698, "clip_ratio/low_mean": 0.09706128179095685, "clip_ratio/low_min": 0.09706128179095685, "clip_ratio/high_mean": 0.04374999925494194, "clip_ratio/high_max": 0.04374999925494194, "clip_ratio/region_mean": 0.1408112810458988, "reward_total_mean": 0.7829693555831909, "reward_meter_mean": 0.695663571357727, "reward_meter_std": 0.2505611181259155, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9904569983482361, "reward_repeat_soft_std": 0.01703684590756893, "reward_judge_quality_mean": 0.7362500429153442, "reward_judge_quality_std": 0.25376805663108826, "reward_total_composite_mean": 0.7829693555831909, "reward_total_composite_std": 0.08598513901233673} {"timestamp_utc": "2026-04-13T00:58:14Z", "mode": "train", "global_step": 1081, "epoch": 0.10858864892014063, "loss": 0.0424, "grad_norm": 13.180346488952637, "learning_rate": 6.7272727272727275e-06, "num_tokens": 1944230.0, "completions/mean_length": 64.75, "completions/min_length": 58.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.75, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.8699319958686829, "rewards/meter/std": 0.1924993395805359, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.953706681728363, "rewards/repeat_soft/std": 0.03438114374876022, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.08166787773370743, "rewards/total_composite/mean": 0.7137150764465332, "rewards/total_composite/std": 0.09152155369520187, "reward": 0.7137150764465332, "reward_std": 0.09152156859636307, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16497564315795898, "sampling/sampling_logp_difference/max": 2.7900748252868652, "sampling/importance_sampling_ratio/min": 0.061416614800691605, "sampling/importance_sampling_ratio/mean": 1.0175936222076416, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2255594059824944, "clip_ratio/low_mean": 0.0509853083640337, "clip_ratio/low_min": 0.0509853083640337, "clip_ratio/high_mean": 0.12242579087615013, "clip_ratio/high_max": 0.12242579087615013, "clip_ratio/region_mean": 0.17341109924018383, "reward_total_mean": 0.7137150764465332, "reward_meter_mean": 0.8699319958686829, "reward_meter_std": 0.1924993395805359, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.953706681728363, "reward_repeat_soft_std": 0.03438114374876022, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.08166787773370743, "reward_total_composite_mean": 0.7137150764465332, "reward_total_composite_std": 0.09152155369520187} {"timestamp_utc": "2026-04-13T00:58:21Z", "mode": "train", "global_step": 1082, "epoch": 0.10868910095429432, "loss": -0.0446, "grad_norm": 15.233142852783203, "learning_rate": 6.724242424242424e-06, "num_tokens": 1945725.0, "completions/mean_length": 38.875, "completions/min_length": 30.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.875, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9669357538223267, "rewards/meter/std": 0.04402047023177147, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9800864458084106, "rewards/repeat_soft/std": 0.01818634197115898, "rewards/judge_quality/mean": 0.5049999952316284, "rewards/judge_quality/std": 0.16801361739635468, "rewards/total_composite/mean": 0.731295108795166, "rewards/total_composite/std": 0.3009405732154846, "reward": 0.731295108795166, "reward_std": 0.3009405732154846, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14369800686836243, "sampling/sampling_logp_difference/max": 1.8150901794433594, "sampling/importance_sampling_ratio/min": 0.16282323002815247, "sampling/importance_sampling_ratio/mean": 0.9896994829177856, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1464561745524406, "clip_ratio/low_mean": 0.02500000037252903, "clip_ratio/low_min": 0.02500000037252903, "clip_ratio/high_mean": 0.154747080989182, "clip_ratio/high_max": 0.154747080989182, "clip_ratio/region_mean": 0.17974708136171103, "reward_total_mean": 0.731295108795166, "reward_meter_mean": 0.9669357538223267, "reward_meter_std": 0.04402047023177147, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9800864458084106, "reward_repeat_soft_std": 0.01818634197115898, "reward_judge_quality_mean": 0.5049999952316284, "reward_judge_quality_std": 0.16801361739635468, "reward_total_composite_mean": 0.731295108795166, "reward_total_composite_std": 0.3009405732154846} {"timestamp_utc": "2026-04-13T00:58:34Z", "mode": "train", "global_step": 1083, "epoch": 0.10878955298844802, "loss": -0.1825, "grad_norm": 1.833428144454956, "learning_rate": 6.721212121212122e-06, "num_tokens": 1947912.0, "completions/mean_length": 135.375, "completions/min_length": 65.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 81.5714340209961, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 89.0, "rewards/meter/mean": 0.9797474145889282, "rewards/meter/std": 0.03441563621163368, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9709240198135376, "rewards/repeat_soft/std": 0.018608249723911285, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.1524970829486847, "rewards/total_composite/mean": 0.6952031850814819, "rewards/total_composite/std": 0.28279992938041687, "reward": 0.6952031850814819, "reward_std": 0.2827998995780945, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12638838589191437, "sampling/sampling_logp_difference/max": 1.6242971420288086, "sampling/importance_sampling_ratio/min": 0.19705013930797577, "sampling/importance_sampling_ratio/mean": 1.0170117616653442, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8163030296564102, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.11887049209326506, "clip_ratio/high_max": 0.11887049209326506, "clip_ratio/region_mean": 0.11887049209326506, "reward_total_mean": 0.6952031850814819, "reward_meter_mean": 0.9797474145889282, "reward_meter_std": 0.03441563621163368, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9709240198135376, "reward_repeat_soft_std": 0.018608249723911285, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.1524970829486847, "reward_total_composite_mean": 0.6952031850814819, "reward_total_composite_std": 0.28279992938041687} {"timestamp_utc": "2026-04-13T00:58:43Z", "mode": "train", "global_step": 1084, "epoch": 0.10889000502260171, "loss": 0.0321, "grad_norm": 10.242932319641113, "learning_rate": 6.718181818181819e-06, "num_tokens": 1950033.0, "completions/mean_length": 92.125, "completions/min_length": 64.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.125, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.7757571339607239, "rewards/meter/std": 0.3116120398044586, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9650493860244751, "rewards/repeat_soft/std": 0.026315871626138687, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.6889706254005432, "rewards/total_composite/std": 0.10859791189432144, "reward": 0.6889706254005432, "reward_std": 0.10859792679548264, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14859648048877716, "sampling/sampling_logp_difference/max": 2.111973762512207, "sampling/importance_sampling_ratio/min": 0.12199617922306061, "sampling/importance_sampling_ratio/mean": 1.0300710201263428, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0202389061450958, "clip_ratio/low_mean": 0.04182238690555096, "clip_ratio/low_min": 0.04182238690555096, "clip_ratio/high_mean": 0.09499135799705982, "clip_ratio/high_max": 0.09499135799705982, "clip_ratio/region_mean": 0.13681374490261078, "reward_total_mean": 0.6889706254005432, "reward_meter_mean": 0.7757571339607239, "reward_meter_std": 0.3116120398044586, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9650493860244751, "reward_repeat_soft_std": 0.026315871626138687, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.6889706254005432, "reward_total_composite_std": 0.10859791189432144} {"timestamp_utc": "2026-04-13T00:58:52Z", "mode": "train", "global_step": 1085, "epoch": 0.1089904570567554, "loss": -0.0574, "grad_norm": 16.87726593017578, "learning_rate": 6.715151515151516e-06, "num_tokens": 1951662.0, "completions/mean_length": 39.625, "completions/min_length": 30.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.625, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.963360071182251, "rewards/meter/std": 0.0580669529736042, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8994936943054199, "rewards/repeat_soft/std": 0.09094548225402832, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7430863976478577, "rewards/total_composite/std": 0.027655774727463722, "reward": 0.7430863976478577, "reward_std": 0.027655763551592827, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1471645087003708, "sampling/sampling_logp_difference/max": 1.5897207260131836, "sampling/importance_sampling_ratio/min": 0.20398256182670593, "sampling/importance_sampling_ratio/mean": 1.0181963443756104, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0946955531835556, "clip_ratio/low_mean": 0.07083333469927311, "clip_ratio/low_min": 0.07083333469927311, "clip_ratio/high_mean": 0.08374990150332451, "clip_ratio/high_max": 0.08374990150332451, "clip_ratio/region_mean": 0.15458323620259762, "reward_total_mean": 0.7430863976478577, "reward_meter_mean": 0.963360071182251, "reward_meter_std": 0.0580669529736042, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8994936943054199, "reward_repeat_soft_std": 0.09094548225402832, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7430863976478577, "reward_total_composite_std": 0.027655774727463722} {"timestamp_utc": "2026-04-13T00:59:00Z", "mode": "train", "global_step": 1086, "epoch": 0.10909090909090909, "loss": 0.0161, "grad_norm": 11.861923217773438, "learning_rate": 6.712121212121213e-06, "num_tokens": 1953582.0, "completions/mean_length": 68.0, "completions/min_length": 56.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.9236580729484558, "rewards/meter/std": 0.051385506987571716, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9892169833183289, "rewards/repeat_soft/std": 0.007870595902204514, "rewards/judge_quality/mean": 0.5024999976158142, "rewards/judge_quality/std": 0.1348809152841568, "rewards/total_composite/mean": 0.7778178453445435, "rewards/total_composite/std": 0.057223767042160034, "reward": 0.7778178453445435, "reward_std": 0.05722375586628914, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16927094757556915, "sampling/sampling_logp_difference/max": 2.152109146118164, "sampling/importance_sampling_ratio/min": 0.11623872816562653, "sampling/importance_sampling_ratio/mean": 1.0093817710876465, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.222341038286686, "clip_ratio/low_mean": 0.10665294737555087, "clip_ratio/low_min": 0.10665294737555087, "clip_ratio/high_mean": 0.041289594024419785, "clip_ratio/high_max": 0.041289594024419785, "clip_ratio/region_mean": 0.14794254139997065, "reward_total_mean": 0.7778178453445435, "reward_meter_mean": 0.9236580729484558, "reward_meter_std": 0.051385506987571716, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9892169833183289, "reward_repeat_soft_std": 0.007870595902204514, "reward_judge_quality_mean": 0.5024999976158142, "reward_judge_quality_std": 0.1348809152841568, "reward_total_composite_mean": 0.7778178453445435, "reward_total_composite_std": 0.057223767042160034} {"timestamp_utc": "2026-04-13T00:59:07Z", "mode": "train", "global_step": 1087, "epoch": 0.10919136112506278, "loss": 0.0532, "grad_norm": 23.66822624206543, "learning_rate": 6.709090909090909e-06, "num_tokens": 1954968.0, "completions/mean_length": 24.25, "completions/min_length": 16.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.25, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.8988249897956848, "rewards/meter/std": 0.23759931325912476, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9517285823822021, "rewards/repeat_soft/std": 0.019798442721366882, "rewards/judge_quality/mean": 0.3474999964237213, "rewards/judge_quality/std": 0.11310551315546036, "rewards/total_composite/mean": 0.7538940906524658, "rewards/total_composite/std": 0.12141013145446777, "reward": 0.7538940906524658, "reward_std": 0.12141013890504837, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15273703634738922, "sampling/sampling_logp_difference/max": 1.0686321258544922, "sampling/importance_sampling_ratio/min": 0.3434780240058899, "sampling/importance_sampling_ratio/mean": 1.0109665393829346, "sampling/importance_sampling_ratio/max": 1.8602601289749146, "entropy": 1.2867448180913925, "clip_ratio/low_mean": 0.02350427396595478, "clip_ratio/low_min": 0.02350427396595478, "clip_ratio/high_mean": 0.11782273650169373, "clip_ratio/high_max": 0.11782273650169373, "clip_ratio/region_mean": 0.1413270104676485, "reward_total_mean": 0.7538940906524658, "reward_meter_mean": 0.8988249897956848, "reward_meter_std": 0.23759931325912476, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9517285823822021, "reward_repeat_soft_std": 0.019798442721366882, "reward_judge_quality_mean": 0.3474999964237213, "reward_judge_quality_std": 0.11310551315546036, "reward_total_composite_mean": 0.7538940906524658, "reward_total_composite_std": 0.12141013145446777} {"timestamp_utc": "2026-04-13T00:59:15Z", "mode": "train", "global_step": 1088, "epoch": 0.10929181315921647, "loss": 0.0153, "grad_norm": 10.723002433776855, "learning_rate": 6.706060606060607e-06, "num_tokens": 1957007.0, "completions/mean_length": 76.875, "completions/min_length": 66.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.875, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.9714881181716919, "rewards/meter/std": 0.044487569481134415, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9344692230224609, "rewards/repeat_soft/std": 0.04513901099562645, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.7878665924072266, "rewards/total_composite/std": 0.04072423651814461, "reward": 0.7878665924072266, "reward_std": 0.040724240243434906, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14112578332424164, "sampling/sampling_logp_difference/max": 2.3885200023651123, "sampling/importance_sampling_ratio/min": 0.091765396296978, "sampling/importance_sampling_ratio/mean": 1.0328235626220703, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9562144950032234, "clip_ratio/low_mean": 0.10142339300364256, "clip_ratio/low_min": 0.10142339300364256, "clip_ratio/high_mean": 0.03032700438052416, "clip_ratio/high_max": 0.03032700438052416, "clip_ratio/region_mean": 0.13175039738416672, "reward_total_mean": 0.7878665924072266, "reward_meter_mean": 0.9714881181716919, "reward_meter_std": 0.044487569481134415, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9344692230224609, "reward_repeat_soft_std": 0.04513901099562645, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.7878665924072266, "reward_total_composite_std": 0.04072423651814461} {"timestamp_utc": "2026-04-13T00:59:22Z", "mode": "train", "global_step": 1089, "epoch": 0.10939226519337017, "loss": 0.1662, "grad_norm": 25.35810089111328, "learning_rate": 6.703030303030304e-06, "num_tokens": 1958644.0, "completions/mean_length": 32.625, "completions/min_length": 23.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.625, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.38219186663627625, "rewards/meter/std": 0.34193259477615356, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9666071534156799, "rewards/repeat_soft/std": 0.03281540796160698, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.478147029876709, "rewards/total_composite/std": 0.1740798056125641, "reward": 0.478147029876709, "reward_std": 0.1740797907114029, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15000052750110626, "sampling/sampling_logp_difference/max": 2.2504515647888184, "sampling/importance_sampling_ratio/min": 0.10535164177417755, "sampling/importance_sampling_ratio/mean": 1.0124231576919556, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8611311465501785, "clip_ratio/low_mean": 0.05535828275606036, "clip_ratio/low_min": 0.05535828275606036, "clip_ratio/high_mean": 0.09463521093130112, "clip_ratio/high_max": 0.09463521093130112, "clip_ratio/region_mean": 0.14999349368736148, "reward_total_mean": 0.478147029876709, "reward_meter_mean": 0.38219186663627625, "reward_meter_std": 0.34193259477615356, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9666071534156799, "reward_repeat_soft_std": 0.03281540796160698, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.478147029876709, "reward_total_composite_std": 0.1740798056125641} {"timestamp_utc": "2026-04-13T00:59:35Z", "mode": "train", "global_step": 1090, "epoch": 0.10949271722752386, "loss": -0.035, "grad_norm": 5.753058910369873, "learning_rate": 6.700000000000001e-06, "num_tokens": 1960080.0, "completions/mean_length": 84.5, "completions/min_length": 22.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 23.428571701049805, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.49229857325553894, "rewards/meter/std": 0.4736401438713074, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.954291582107544, "rewards/repeat_soft/std": 0.023216886445879936, "rewards/judge_quality/mean": 0.3475000262260437, "rewards/judge_quality/std": 0.15563234686851501, "rewards/total_composite/mean": 0.4562535881996155, "rewards/total_composite/std": 0.3413672149181366, "reward": 0.4562535881996155, "reward_std": 0.341367244720459, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.190608412027359, "sampling/sampling_logp_difference/max": 1.1815366744995117, "sampling/importance_sampling_ratio/min": 0.30680692195892334, "sampling/importance_sampling_ratio/mean": 1.0208443403244019, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2288463488221169, "clip_ratio/low_mean": 0.07998251914978027, "clip_ratio/low_min": 0.07998251914978027, "clip_ratio/high_mean": 0.07548701576888561, "clip_ratio/high_max": 0.07548701576888561, "clip_ratio/region_mean": 0.15546953491866589, "reward_total_mean": 0.4562535881996155, "reward_meter_mean": 0.49229857325553894, "reward_meter_std": 0.4736401438713074, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.954291582107544, "reward_repeat_soft_std": 0.023216886445879936, "reward_judge_quality_mean": 0.3475000262260437, "reward_judge_quality_std": 0.15563234686851501, "reward_total_composite_mean": 0.4562535881996155, "reward_total_composite_std": 0.3413672149181366} {"timestamp_utc": "2026-04-13T00:59:48Z", "mode": "train", "global_step": 1091, "epoch": 0.10959316926167754, "loss": -0.0957, "grad_norm": 2.3743202686309814, "learning_rate": 6.6969696969696975e-06, "num_tokens": 1961687.0, "completions/mean_length": 91.875, "completions/min_length": 26.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 31.85714340209961, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.668614387512207, "rewards/meter/std": 0.3471478521823883, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9965097904205322, "rewards/repeat_soft/std": 0.004329604562371969, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.18031717836856842, "rewards/total_composite/mean": 0.6419024467468262, "rewards/total_composite/std": 0.2818882465362549, "reward": 0.6419024467468262, "reward_std": 0.2818882167339325, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1278303563594818, "sampling/sampling_logp_difference/max": 1.1583795547485352, "sampling/importance_sampling_ratio/min": 0.3139945864677429, "sampling/importance_sampling_ratio/mean": 1.0093224048614502, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7205618917942047, "clip_ratio/low_mean": 0.037913603708148, "clip_ratio/low_min": 0.037913603708148, "clip_ratio/high_mean": 0.08342884946614504, "clip_ratio/high_max": 0.08342884946614504, "clip_ratio/region_mean": 0.12134245317429304, "reward_total_mean": 0.6419024467468262, "reward_meter_mean": 0.668614387512207, "reward_meter_std": 0.3471478521823883, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9965097904205322, "reward_repeat_soft_std": 0.004329604562371969, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.18031717836856842, "reward_total_composite_mean": 0.6419024467468262, "reward_total_composite_std": 0.2818882465362549} {"timestamp_utc": "2026-04-13T00:59:55Z", "mode": "train", "global_step": 1092, "epoch": 0.10969362129583124, "loss": 0.0664, "grad_norm": 12.07098388671875, "learning_rate": 6.693939393939395e-06, "num_tokens": 1963507.0, "completions/mean_length": 55.5, "completions/min_length": 50.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.5, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.984637975692749, "rewards/meter/std": 0.009531737305223942, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8852827548980713, "rewards/repeat_soft/std": 0.16873621940612793, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.7321153283119202, "rewards/total_composite/std": 0.03166723623871803, "reward": 0.7321153283119202, "reward_std": 0.03166722133755684, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12863391637802124, "sampling/sampling_logp_difference/max": 1.6166057586669922, "sampling/importance_sampling_ratio/min": 0.19857154786586761, "sampling/importance_sampling_ratio/mean": 1.0253863334655762, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9659101068973541, "clip_ratio/low_mean": 0.06079545570537448, "clip_ratio/low_min": 0.06079545570537448, "clip_ratio/high_mean": 0.05572440195828676, "clip_ratio/high_max": 0.05572440195828676, "clip_ratio/region_mean": 0.11651985766366124, "reward_total_mean": 0.7321153283119202, "reward_meter_mean": 0.984637975692749, "reward_meter_std": 0.009531737305223942, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8852827548980713, "reward_repeat_soft_std": 0.16873621940612793, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.7321153283119202, "reward_total_composite_std": 0.03166723623871803} {"timestamp_utc": "2026-04-13T01:00:04Z", "mode": "train", "global_step": 1093, "epoch": 0.10979407332998493, "loss": -0.0895, "grad_norm": 7.840693473815918, "learning_rate": 6.690909090909091e-06, "num_tokens": 1965579.0, "completions/mean_length": 83.0, "completions/min_length": 62.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.9847190380096436, "rewards/meter/std": 0.007758138235658407, "rewards/count_adherence/mean": 0.6750000715255737, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9269916415214539, "rewards/repeat_soft/std": 0.04161645844578743, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.7679476737976074, "rewards/total_composite/std": 0.04609550163149834, "reward": 0.7679476737976074, "reward_std": 0.04609549045562744, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1369951069355011, "sampling/sampling_logp_difference/max": 2.2820377349853516, "sampling/importance_sampling_ratio/min": 0.1020759865641594, "sampling/importance_sampling_ratio/mean": 1.0105382204055786, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9167511388659477, "clip_ratio/low_mean": 0.06798156350851059, "clip_ratio/low_min": 0.06798156350851059, "clip_ratio/high_mean": 0.06165452487766743, "clip_ratio/high_max": 0.06165452487766743, "clip_ratio/region_mean": 0.12963608838617802, "reward_total_mean": 0.7679476737976074, "reward_meter_mean": 0.9847190380096436, "reward_meter_std": 0.007758138235658407, "reward_count_adherence_mean": 0.6750000715255737, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9269916415214539, "reward_repeat_soft_std": 0.04161645844578743, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.7679476737976074, "reward_total_composite_std": 0.04609550163149834} {"timestamp_utc": "2026-04-13T01:00:11Z", "mode": "train", "global_step": 1094, "epoch": 0.10989452536413863, "loss": 0.0389, "grad_norm": 19.0589542388916, "learning_rate": 6.687878787878788e-06, "num_tokens": 1966990.0, "completions/mean_length": 33.375, "completions/min_length": 29.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.375, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.6641157865524292, "rewards/meter/std": 0.41760846972465515, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9942686557769775, "rewards/repeat_soft/std": 0.010715479962527752, "rewards/judge_quality/mean": 0.8575000166893005, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.805528998374939, "rewards/total_composite/std": 0.22453705966472626, "reward": 0.805528998374939, "reward_std": 0.22453705966472626, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13472184538841248, "sampling/sampling_logp_difference/max": 1.402099609375, "sampling/importance_sampling_ratio/min": 0.24607975780963898, "sampling/importance_sampling_ratio/mean": 0.9936306476593018, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7683432102203369, "clip_ratio/low_mean": 0.03639846853911877, "clip_ratio/low_min": 0.03639846853911877, "clip_ratio/high_mean": 0.056635601446032524, "clip_ratio/high_max": 0.056635601446032524, "clip_ratio/region_mean": 0.09303406998515129, "reward_total_mean": 0.805528998374939, "reward_meter_mean": 0.6641157865524292, "reward_meter_std": 0.41760846972465515, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9942686557769775, "reward_repeat_soft_std": 0.010715479962527752, "reward_judge_quality_mean": 0.8575000166893005, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.805528998374939, "reward_total_composite_std": 0.22453705966472626} {"timestamp_utc": "2026-04-13T01:00:19Z", "mode": "train", "global_step": 1095, "epoch": 0.10999497739829231, "loss": 0.0513, "grad_norm": 14.06030559539795, "learning_rate": 6.684848484848485e-06, "num_tokens": 1968669.0, "completions/mean_length": 47.875, "completions/min_length": 42.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.875, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.9863992929458618, "rewards/meter/std": 0.013339622877538204, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9715620279312134, "rewards/repeat_soft/std": 0.021549014374613762, "rewards/judge_quality/mean": 0.39249998331069946, "rewards/judge_quality/std": 0.08892211318016052, "rewards/total_composite/mean": 0.8087859153747559, "rewards/total_composite/std": 0.02736634761095047, "reward": 0.8087859153747559, "reward_std": 0.02736634761095047, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17555293440818787, "sampling/sampling_logp_difference/max": 1.5651001930236816, "sampling/importance_sampling_ratio/min": 0.20906706154346466, "sampling/importance_sampling_ratio/mean": 1.032060980796814, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.171095073223114, "clip_ratio/low_mean": 0.0421899538487196, "clip_ratio/low_min": 0.0421899538487196, "clip_ratio/high_mean": 0.1496925838291645, "clip_ratio/high_max": 0.1496925838291645, "clip_ratio/region_mean": 0.1918825376778841, "reward_total_mean": 0.8087859153747559, "reward_meter_mean": 0.9863992929458618, "reward_meter_std": 0.013339622877538204, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9715620279312134, "reward_repeat_soft_std": 0.021549014374613762, "reward_judge_quality_mean": 0.39249998331069946, "reward_judge_quality_std": 0.08892211318016052, "reward_total_composite_mean": 0.8087859153747559, "reward_total_composite_std": 0.02736634761095047} {"timestamp_utc": "2026-04-13T01:00:26Z", "mode": "train", "global_step": 1096, "epoch": 0.110095429432446, "loss": 0.069, "grad_norm": 16.348186492919922, "learning_rate": 6.681818181818183e-06, "num_tokens": 1970576.0, "completions/mean_length": 68.375, "completions/min_length": 59.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.375, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.8150389194488525, "rewards/meter/std": 0.26007986068725586, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9769142866134644, "rewards/repeat_soft/std": 0.02559221163392067, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7029589414596558, "rewards/total_composite/std": 0.11708447337150574, "reward": 0.7029589414596558, "reward_std": 0.11708448082208633, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14262130856513977, "sampling/sampling_logp_difference/max": 1.8344573974609375, "sampling/importance_sampling_ratio/min": 0.15970014035701752, "sampling/importance_sampling_ratio/mean": 1.0055125951766968, "sampling/importance_sampling_ratio/max": 1.9072970151901245, "entropy": 0.971847653388977, "clip_ratio/low_mean": 0.03643822483718395, "clip_ratio/low_min": 0.03643822483718395, "clip_ratio/high_mean": 0.09220616612583399, "clip_ratio/high_max": 0.09220616612583399, "clip_ratio/region_mean": 0.12864439096301794, "reward_total_mean": 0.7029589414596558, "reward_meter_mean": 0.8150389194488525, "reward_meter_std": 0.26007986068725586, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9769142866134644, "reward_repeat_soft_std": 0.02559221163392067, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7029589414596558, "reward_total_composite_std": 0.11708447337150574} {"timestamp_utc": "2026-04-13T01:00:33Z", "mode": "train", "global_step": 1097, "epoch": 0.1101958814665997, "loss": 0.0266, "grad_norm": 14.547167778015137, "learning_rate": 6.678787878787879e-06, "num_tokens": 1972091.0, "completions/mean_length": 28.375, "completions/min_length": 23.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.375, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9194705486297607, "rewards/meter/std": 0.21812871098518372, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9520623683929443, "rewards/repeat_soft/std": 0.016888923943042755, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.7557179927825928, "rewards/total_composite/std": 0.09175864607095718, "reward": 0.7557179927825928, "reward_std": 0.09175864607095718, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12418671697378159, "sampling/sampling_logp_difference/max": 1.3031902313232422, "sampling/importance_sampling_ratio/min": 0.27166375517845154, "sampling/importance_sampling_ratio/mean": 1.015838623046875, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9235247150063515, "clip_ratio/low_mean": 0.013141026254743338, "clip_ratio/low_min": 0.013141026254743338, "clip_ratio/high_mean": 0.11643166095018387, "clip_ratio/high_max": 0.11643166095018387, "clip_ratio/region_mean": 0.1295726872049272, "reward_total_mean": 0.7557179927825928, "reward_meter_mean": 0.9194705486297607, "reward_meter_std": 0.21812871098518372, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9520623683929443, "reward_repeat_soft_std": 0.016888923943042755, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.7557179927825928, "reward_total_composite_std": 0.09175864607095718} {"timestamp_utc": "2026-04-13T01:00:40Z", "mode": "train", "global_step": 1098, "epoch": 0.1102963335007534, "loss": 0.0307, "grad_norm": 12.453449249267578, "learning_rate": 6.6757575757575766e-06, "num_tokens": 1973780.0, "completions/mean_length": 52.125, "completions/min_length": 47.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.125, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.753472089767456, "rewards/meter/std": 0.4050270617008209, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9420626163482666, "rewards/repeat_soft/std": 0.06425425410270691, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.20119288563728333, "rewards/total_composite/mean": 0.7160186767578125, "rewards/total_composite/std": 0.1641184389591217, "reward": 0.7160186767578125, "reward_std": 0.1641184538602829, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13318811357021332, "sampling/sampling_logp_difference/max": 1.03718900680542, "sampling/importance_sampling_ratio/min": 0.35444965958595276, "sampling/importance_sampling_ratio/mean": 1.0071898698806763, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9363749027252197, "clip_ratio/low_mean": 0.0325681222602725, "clip_ratio/low_min": 0.0325681222602725, "clip_ratio/high_mean": 0.10401653219014406, "clip_ratio/high_max": 0.10401653219014406, "clip_ratio/region_mean": 0.13658465445041656, "reward_total_mean": 0.7160186767578125, "reward_meter_mean": 0.753472089767456, "reward_meter_std": 0.4050270617008209, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9420626163482666, "reward_repeat_soft_std": 0.06425425410270691, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.20119288563728333, "reward_total_composite_mean": 0.7160186767578125, "reward_total_composite_std": 0.1641184389591217} {"timestamp_utc": "2026-04-13T01:00:48Z", "mode": "train", "global_step": 1099, "epoch": 0.11039678553490709, "loss": 0.0187, "grad_norm": 8.995245933532715, "learning_rate": 6.672727272727273e-06, "num_tokens": 1976049.0, "completions/mean_length": 86.625, "completions/min_length": 74.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.625, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.9686621427536011, "rewards/meter/std": 0.017199097201228142, "rewards/count_adherence/mean": 0.7749999761581421, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.919672966003418, "rewards/repeat_soft/std": 0.08524598181247711, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.7813652753829956, "rewards/total_composite/std": 0.0385681577026844, "reward": 0.7813652753829956, "reward_std": 0.03856814652681351, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1422412097454071, "sampling/sampling_logp_difference/max": 1.7741403579711914, "sampling/importance_sampling_ratio/min": 0.16962921619415283, "sampling/importance_sampling_ratio/mean": 1.0212472677230835, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9633924588561058, "clip_ratio/low_mean": 0.08459246437996626, "clip_ratio/low_min": 0.08459246437996626, "clip_ratio/high_mean": 0.04927306156605482, "clip_ratio/high_max": 0.04927306156605482, "clip_ratio/region_mean": 0.13386552594602108, "reward_total_mean": 0.7813652753829956, "reward_meter_mean": 0.9686621427536011, "reward_meter_std": 0.017199097201228142, "reward_count_adherence_mean": 0.7749999761581421, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.919672966003418, "reward_repeat_soft_std": 0.08524598181247711, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.7813652753829956, "reward_total_composite_std": 0.0385681577026844} {"timestamp_utc": "2026-04-13T01:00:56Z", "mode": "train", "global_step": 1100, "epoch": 0.11049723756906077, "loss": 0.0111, "grad_norm": 11.893102645874023, "learning_rate": 6.66969696969697e-06, "num_tokens": 1978048.0, "completions/mean_length": 68.875, "completions/min_length": 62.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.875, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.8619036078453064, "rewards/meter/std": 0.2469317466020584, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9546021819114685, "rewards/repeat_soft/std": 0.031168954446911812, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.7405668497085571, "rewards/total_composite/std": 0.1327940821647644, "reward": 0.7405668497085571, "reward_std": 0.1327940821647644, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1398290991783142, "sampling/sampling_logp_difference/max": 1.3374385833740234, "sampling/importance_sampling_ratio/min": 0.26251721382141113, "sampling/importance_sampling_ratio/mean": 1.0202773809432983, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.988353781402111, "clip_ratio/low_mean": 0.03365384694188833, "clip_ratio/low_min": 0.03365384694188833, "clip_ratio/high_mean": 0.09163326863199472, "clip_ratio/high_max": 0.09163326863199472, "clip_ratio/region_mean": 0.12528711557388306, "reward_total_mean": 0.7405668497085571, "reward_meter_mean": 0.8619036078453064, "reward_meter_std": 0.2469317466020584, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9546021819114685, "reward_repeat_soft_std": 0.031168954446911812, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.7405668497085571, "reward_total_composite_std": 0.1327940821647644} {"timestamp_utc": "2026-04-13T01:02:11Z", "mode": "eval", "global_step": 1100, "epoch": 0.11049723756906077, "eval_loss": NaN, "eval_runtime": 74.6024, "eval_samples_per_second": 1.072, "eval_steps_per_second": 0.134, "eval_num_tokens": 1978048.0, "eval_completions/mean_length": 87.75, "eval_completions/min_length": 30.6, "eval_completions/max_length": 298.5, "eval_completions/clipped_ratio": 0.0625, "eval_completions/mean_terminated_length": 59.97678756713867, "eval_completions/min_terminated_length": 30.6, "eval_completions/max_terminated_length": 103.8, "eval_rewards/meter/mean": 0.766729611158371, "eval_rewards/meter/std": 0.3216488525271416, "eval_rewards/count_adherence/mean": 0.7912500023841857, "eval_rewards/count_adherence/std": 0.1757865823805332, "eval_rewards/hard_gate/mean": 0.925, "eval_rewards/hard_gate/std": 0.18771235942840575, "eval_rewards/repeat_soft/mean": 0.9521731436252594, "eval_rewards/repeat_soft/std": 0.05752834491431713, "eval_rewards/judge_quality/mean": 0.4430000066757202, "eval_rewards/judge_quality/std": 0.16298070801422, "eval_rewards/total_composite/mean": 0.6643548518419266, "eval_rewards/total_composite/std": 0.22286631651222705, "eval_reward": 0.6643548518419266, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.08288106843829154, "eval_sampling/sampling_logp_difference/max": 0.9737040042877197, "eval_sampling/importance_sampling_ratio/min": 0.394601945579052, "eval_sampling/importance_sampling_ratio/mean": 1.0233500003814697, "eval_sampling/importance_sampling_ratio/max": 1.5098692893981933, "eval_entropy": 0.9797285854816437, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6643548518419266, "eval_reward_meter_mean": 0.766729611158371, "eval_reward_meter_std": 0.3216488525271416, "eval_reward_count_adherence_mean": 0.7912500023841857, "eval_reward_count_adherence_std": 0.1757865823805332, "eval_reward_hard_gate_mean": 0.925, "eval_reward_hard_gate_std": 0.18771235942840575, "eval_reward_repeat_soft_mean": 0.9521731436252594, "eval_reward_repeat_soft_std": 0.05752834491431713, "eval_reward_judge_quality_mean": 0.4430000066757202, "eval_reward_judge_quality_std": 0.16298070801422, "eval_reward_total_composite_mean": 0.6643548518419266, "eval_reward_total_composite_std": 0.22286631651222705} {"timestamp_utc": "2026-04-13T01:02:21Z", "mode": "train", "global_step": 1101, "epoch": 0.11059768960321446, "loss": 0.03, "grad_norm": 12.044756889343262, "learning_rate": 6.666666666666667e-06, "num_tokens": 1979792.0, "completions/mean_length": 56.0, "completions/min_length": 50.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.0, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9946514368057251, "rewards/meter/std": 0.00222483417019248, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9962372183799744, "rewards/repeat_soft/std": 0.004324719309806824, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.8442168831825256, "rewards/total_composite/std": 0.05151517689228058, "reward": 0.8442168831825256, "reward_std": 0.05151517316699028, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15390346944332123, "sampling/sampling_logp_difference/max": 1.7719106674194336, "sampling/importance_sampling_ratio/min": 0.1700078547000885, "sampling/importance_sampling_ratio/mean": 1.0372693538665771, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3155619576573372, "clip_ratio/low_mean": 0.09879409987479448, "clip_ratio/low_min": 0.09879409987479448, "clip_ratio/high_mean": 0.0245535708963871, "clip_ratio/high_max": 0.0245535708963871, "clip_ratio/region_mean": 0.12334767077118158, "reward_total_mean": 0.8442168831825256, "reward_meter_mean": 0.9946514368057251, "reward_meter_std": 0.00222483417019248, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9962372183799744, "reward_repeat_soft_std": 0.004324719309806824, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.8442168831825256, "reward_total_composite_std": 0.05151517689228058} {"timestamp_utc": "2026-04-13T01:02:29Z", "mode": "train", "global_step": 1102, "epoch": 0.11069814163736816, "loss": 0.0739, "grad_norm": 11.022254943847656, "learning_rate": 6.663636363636365e-06, "num_tokens": 1982109.0, "completions/mean_length": 77.625, "completions/min_length": 65.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.625, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.9513521790504456, "rewards/meter/std": 0.08910845220088959, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9525443911552429, "rewards/repeat_soft/std": 0.03233771026134491, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.7817379236221313, "rewards/total_composite/std": 0.07238435745239258, "reward": 0.7817379236221313, "reward_std": 0.07238434255123138, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12863284349441528, "sampling/sampling_logp_difference/max": 1.8511686325073242, "sampling/importance_sampling_ratio/min": 0.15705353021621704, "sampling/importance_sampling_ratio/mean": 0.9995982050895691, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7204125076532364, "clip_ratio/low_mean": 0.08676480781286955, "clip_ratio/low_min": 0.08676480781286955, "clip_ratio/high_mean": 0.030300221405923367, "clip_ratio/high_max": 0.030300221405923367, "clip_ratio/region_mean": 0.11706502921879292, "reward_total_mean": 0.7817379236221313, "reward_meter_mean": 0.9513521790504456, "reward_meter_std": 0.08910845220088959, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9525443911552429, "reward_repeat_soft_std": 0.03233771026134491, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.7817379236221313, "reward_total_composite_std": 0.07238435745239258} {"timestamp_utc": "2026-04-13T01:02:38Z", "mode": "train", "global_step": 1103, "epoch": 0.11079859367152185, "loss": 0.0223, "grad_norm": 7.148165225982666, "learning_rate": 6.660606060606061e-06, "num_tokens": 1984117.0, "completions/mean_length": 79.0, "completions/min_length": 57.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.0, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9878445863723755, "rewards/meter/std": 0.011258679442107677, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9375649690628052, "rewards/repeat_soft/std": 0.05307085067033768, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7704115509986877, "rewards/total_composite/std": 0.017845269292593002, "reward": 0.7704115509986877, "reward_std": 0.017845269292593002, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12757059931755066, "sampling/sampling_logp_difference/max": 1.3445243835449219, "sampling/importance_sampling_ratio/min": 0.2606636583805084, "sampling/importance_sampling_ratio/mean": 1.0067778825759888, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0541097074747086, "clip_ratio/low_mean": 0.03725058399140835, "clip_ratio/low_min": 0.03725058399140835, "clip_ratio/high_mean": 0.07628756342455745, "clip_ratio/high_max": 0.07628756342455745, "clip_ratio/region_mean": 0.1135381474159658, "reward_total_mean": 0.7704115509986877, "reward_meter_mean": 0.9878445863723755, "reward_meter_std": 0.011258679442107677, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9375649690628052, "reward_repeat_soft_std": 0.05307085067033768, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7704115509986877, "reward_total_composite_std": 0.017845269292593002} {"timestamp_utc": "2026-04-13T01:02:45Z", "mode": "train", "global_step": 1104, "epoch": 0.11089904570567553, "loss": 0.0122, "grad_norm": 14.85375690460205, "learning_rate": 6.657575757575758e-06, "num_tokens": 1985721.0, "completions/mean_length": 41.5, "completions/min_length": 39.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.5, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.8686255812644958, "rewards/meter/std": 0.22695152461528778, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8972636461257935, "rewards/repeat_soft/std": 0.2090158313512802, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.751357913017273, "rewards/total_composite/std": 0.10411975532770157, "reward": 0.751357913017273, "reward_std": 0.10411974787712097, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1532209813594818, "sampling/sampling_logp_difference/max": 2.2931718826293945, "sampling/importance_sampling_ratio/min": 0.10094577074050903, "sampling/importance_sampling_ratio/mean": 1.0102002620697021, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.877097986638546, "clip_ratio/low_mean": 0.015109890373423696, "clip_ratio/low_min": 0.015109890373423696, "clip_ratio/high_mean": 0.13860789965838194, "clip_ratio/high_max": 0.13860789965838194, "clip_ratio/region_mean": 0.15371779003180563, "reward_total_mean": 0.751357913017273, "reward_meter_mean": 0.8686255812644958, "reward_meter_std": 0.22695152461528778, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8972636461257935, "reward_repeat_soft_std": 0.2090158313512802, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.751357913017273, "reward_total_composite_std": 0.10411975532770157} {"timestamp_utc": "2026-04-13T01:02:53Z", "mode": "train", "global_step": 1105, "epoch": 0.11099949773982923, "loss": 0.0106, "grad_norm": 8.506380081176758, "learning_rate": 6.654545454545455e-06, "num_tokens": 1987568.0, "completions/mean_length": 78.875, "completions/min_length": 57.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.875, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.9853217601776123, "rewards/meter/std": 0.003925495781004429, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8184282779693604, "rewards/repeat_soft/std": 0.10667596012353897, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.7937376499176025, "rewards/total_composite/std": 0.04903680086135864, "reward": 0.7937376499176025, "reward_std": 0.04903681203722954, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1182798221707344, "sampling/sampling_logp_difference/max": 1.6223468780517578, "sampling/importance_sampling_ratio/min": 0.19743479788303375, "sampling/importance_sampling_ratio/mean": 0.9957869648933411, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8493968397378922, "clip_ratio/low_mean": 0.058222873602062464, "clip_ratio/low_min": 0.058222873602062464, "clip_ratio/high_mean": 0.027784214355051517, "clip_ratio/high_max": 0.027784214355051517, "clip_ratio/region_mean": 0.08600708795711398, "reward_total_mean": 0.7937376499176025, "reward_meter_mean": 0.9853217601776123, "reward_meter_std": 0.003925495781004429, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8184282779693604, "reward_repeat_soft_std": 0.10667596012353897, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.7937376499176025, "reward_total_composite_std": 0.04903680086135864} {"timestamp_utc": "2026-04-13T01:03:00Z", "mode": "train", "global_step": 1106, "epoch": 0.11109994977398292, "loss": -0.0306, "grad_norm": 14.011372566223145, "learning_rate": 6.651515151515152e-06, "num_tokens": 1989232.0, "completions/mean_length": 34.0, "completions/min_length": 26.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.0, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.9955812096595764, "rewards/meter/std": 0.004840184468775988, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9383153915405273, "rewards/repeat_soft/std": 0.0518212616443634, "rewards/judge_quality/mean": 0.3725000023841858, "rewards/judge_quality/std": 0.11055056750774384, "rewards/total_composite/mean": 0.803593099117279, "rewards/total_composite/std": 0.03321225568652153, "reward": 0.803593099117279, "reward_std": 0.03321225568652153, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1460893601179123, "sampling/sampling_logp_difference/max": 1.0981006622314453, "sampling/importance_sampling_ratio/min": 0.3335039019584656, "sampling/importance_sampling_ratio/mean": 1.0223891735076904, "sampling/importance_sampling_ratio/max": 1.9079362154006958, "entropy": 1.040138691663742, "clip_ratio/low_mean": 0.03585164900869131, "clip_ratio/low_min": 0.03585164900869131, "clip_ratio/high_mean": 0.1418213676661253, "clip_ratio/high_max": 0.1418213676661253, "clip_ratio/region_mean": 0.1776730166748166, "reward_total_mean": 0.803593099117279, "reward_meter_mean": 0.9955812096595764, "reward_meter_std": 0.004840184468775988, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9383153915405273, "reward_repeat_soft_std": 0.0518212616443634, "reward_judge_quality_mean": 0.3725000023841858, "reward_judge_quality_std": 0.11055056750774384, "reward_total_composite_mean": 0.803593099117279, "reward_total_composite_std": 0.03321225568652153} {"timestamp_utc": "2026-04-13T01:03:08Z", "mode": "train", "global_step": 1107, "epoch": 0.11120040180813662, "loss": 0.0195, "grad_norm": 14.347731590270996, "learning_rate": 6.6484848484848485e-06, "num_tokens": 1990928.0, "completions/mean_length": 48.0, "completions/min_length": 42.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.0, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.8265170454978943, "rewards/meter/std": 0.23672540485858917, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8799217343330383, "rewards/repeat_soft/std": 0.08597038686275482, "rewards/judge_quality/mean": 0.42249998450279236, "rewards/judge_quality/std": 0.14616528153419495, "rewards/total_composite/mean": 0.7366748452186584, "rewards/total_composite/std": 0.10457415878772736, "reward": 0.7366748452186584, "reward_std": 0.10457415878772736, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12905117869377136, "sampling/sampling_logp_difference/max": 1.8504302501678467, "sampling/importance_sampling_ratio/min": 0.15716953575611115, "sampling/importance_sampling_ratio/mean": 1.0057438611984253, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8476065620779991, "clip_ratio/low_mean": 0.013181818183511496, "clip_ratio/low_min": 0.013181818183511496, "clip_ratio/high_mean": 0.07243162579834461, "clip_ratio/high_max": 0.07243162579834461, "clip_ratio/region_mean": 0.08561344398185611, "reward_total_mean": 0.7366748452186584, "reward_meter_mean": 0.8265170454978943, "reward_meter_std": 0.23672540485858917, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8799217343330383, "reward_repeat_soft_std": 0.08597038686275482, "reward_judge_quality_mean": 0.42249998450279236, "reward_judge_quality_std": 0.14616528153419495, "reward_total_composite_mean": 0.7366748452186584, "reward_total_composite_std": 0.10457415878772736} {"timestamp_utc": "2026-04-13T01:03:15Z", "mode": "train", "global_step": 1108, "epoch": 0.11130085384229031, "loss": 0.1026, "grad_norm": 25.330612182617188, "learning_rate": 6.645454545454546e-06, "num_tokens": 1992521.0, "completions/mean_length": 36.125, "completions/min_length": 30.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.125, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.6671162843704224, "rewards/meter/std": 0.4045107960700989, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9588849544525146, "rewards/repeat_soft/std": 0.08995931595563889, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.25150617957115173, "rewards/total_composite/mean": 0.653215765953064, "rewards/total_composite/std": 0.17775842547416687, "reward": 0.653215765953064, "reward_std": 0.17775841057300568, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13881176710128784, "sampling/sampling_logp_difference/max": 1.01676607131958, "sampling/importance_sampling_ratio/min": 0.36176297068595886, "sampling/importance_sampling_ratio/mean": 1.025588870048523, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.824974499642849, "clip_ratio/low_mean": 0.0324730109423399, "clip_ratio/low_min": 0.0324730109423399, "clip_ratio/high_mean": 0.108653848990798, "clip_ratio/high_max": 0.108653848990798, "clip_ratio/region_mean": 0.1411268599331379, "reward_total_mean": 0.653215765953064, "reward_meter_mean": 0.6671162843704224, "reward_meter_std": 0.4045107960700989, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9588849544525146, "reward_repeat_soft_std": 0.08995931595563889, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.25150617957115173, "reward_total_composite_mean": 0.653215765953064, "reward_total_composite_std": 0.17775842547416687} {"timestamp_utc": "2026-04-13T01:03:27Z", "mode": "train", "global_step": 1109, "epoch": 0.11140130587644399, "loss": -0.1145, "grad_norm": 2.8939380645751953, "learning_rate": 6.642424242424242e-06, "num_tokens": 1994325.0, "completions/mean_length": 174.5, "completions/min_length": 54.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.21591435372829437, "rewards/meter/std": 0.3108412027359009, "rewards/count_adherence/mean": 0.65625, "rewards/count_adherence/std": 0.18600596487522125, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9812561273574829, "rewards/repeat_soft/std": 0.012569981627166271, "rewards/judge_quality/mean": 0.2574999928474426, "rewards/judge_quality/std": 0.1642080396413803, "rewards/total_composite/mean": 0.3265325427055359, "rewards/total_composite/std": 0.23402124643325806, "reward": 0.3265325427055359, "reward_std": 0.23402124643325806, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16088490188121796, "sampling/sampling_logp_difference/max": 1.4672727584838867, "sampling/importance_sampling_ratio/min": 0.23055340349674225, "sampling/importance_sampling_ratio/mean": 1.0278958082199097, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1153929084539413, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.11543166171759367, "clip_ratio/high_max": 0.11543166171759367, "clip_ratio/region_mean": 0.11543166171759367, "reward_total_mean": 0.3265325427055359, "reward_meter_mean": 0.21591435372829437, "reward_meter_std": 0.3108412027359009, "reward_count_adherence_mean": 0.65625, "reward_count_adherence_std": 0.18600596487522125, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9812561273574829, "reward_repeat_soft_std": 0.012569981627166271, "reward_judge_quality_mean": 0.2574999928474426, "reward_judge_quality_std": 0.1642080396413803, "reward_total_composite_mean": 0.3265325427055359, "reward_total_composite_std": 0.23402124643325806} {"timestamp_utc": "2026-04-13T01:03:35Z", "mode": "train", "global_step": 1110, "epoch": 0.11150175791059769, "loss": -0.0058, "grad_norm": 8.83658504486084, "learning_rate": 6.63939393939394e-06, "num_tokens": 1996528.0, "completions/mean_length": 81.375, "completions/min_length": 68.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.375, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9933127760887146, "rewards/meter/std": 0.004225994925945997, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9574509859085083, "rewards/repeat_soft/std": 0.0358353815972805, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.7861108779907227, "rewards/total_composite/std": 0.03975406289100647, "reward": 0.7861108779907227, "reward_std": 0.03975406289100647, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1218012124300003, "sampling/sampling_logp_difference/max": 1.272587776184082, "sampling/importance_sampling_ratio/min": 0.2801058292388916, "sampling/importance_sampling_ratio/mean": 1.016148328781128, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9570276215672493, "clip_ratio/low_mean": 0.10354544129222631, "clip_ratio/low_min": 0.10354544129222631, "clip_ratio/high_mean": 0.021341463550925255, "clip_ratio/high_max": 0.021341463550925255, "clip_ratio/region_mean": 0.12488690484315157, "reward_total_mean": 0.7861108779907227, "reward_meter_mean": 0.9933127760887146, "reward_meter_std": 0.004225994925945997, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9574509859085083, "reward_repeat_soft_std": 0.0358353815972805, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.7861108779907227, "reward_total_composite_std": 0.03975406289100647} {"timestamp_utc": "2026-04-13T01:03:42Z", "mode": "train", "global_step": 1111, "epoch": 0.11160220994475138, "loss": 0.0169, "grad_norm": 15.736329078674316, "learning_rate": 6.6363636363636375e-06, "num_tokens": 1998178.0, "completions/mean_length": 45.25, "completions/min_length": 39.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.25, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.8965341448783875, "rewards/meter/std": 0.2566237151622772, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9315600991249084, "rewards/repeat_soft/std": 0.047201503068208694, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7534713745117188, "rewards/total_composite/std": 0.13083547353744507, "reward": 0.7534713745117188, "reward_std": 0.13083547353744507, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15885913372039795, "sampling/sampling_logp_difference/max": 2.3007423877716064, "sampling/importance_sampling_ratio/min": 0.10018444061279297, "sampling/importance_sampling_ratio/mean": 1.0230045318603516, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9434115663170815, "clip_ratio/low_mean": 0.019021738320589066, "clip_ratio/low_min": 0.019021738320589066, "clip_ratio/high_mean": 0.12053323606960475, "clip_ratio/high_max": 0.12053323606960475, "clip_ratio/region_mean": 0.13955497439019382, "reward_total_mean": 0.7534713745117188, "reward_meter_mean": 0.8965341448783875, "reward_meter_std": 0.2566237151622772, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9315600991249084, "reward_repeat_soft_std": 0.047201503068208694, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7534713745117188, "reward_total_composite_std": 0.13083547353744507} {"timestamp_utc": "2026-04-13T01:03:51Z", "mode": "train", "global_step": 1112, "epoch": 0.11170266197890508, "loss": 0.1332, "grad_norm": 9.519704818725586, "learning_rate": 6.633333333333334e-06, "num_tokens": 2000458.0, "completions/mean_length": 97.0, "completions/min_length": 68.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.0, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.7248070240020752, "rewards/meter/std": 0.3550230860710144, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8007067441940308, "rewards/repeat_soft/std": 0.10040460526943207, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.12631450593471527, "rewards/total_composite/mean": 0.6181088089942932, "rewards/total_composite/std": 0.17873498797416687, "reward": 0.6181088089942932, "reward_std": 0.17873497307300568, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1386922001838684, "sampling/sampling_logp_difference/max": 2.612384557723999, "sampling/importance_sampling_ratio/min": 0.0733594074845314, "sampling/importance_sampling_ratio/mean": 0.9993525743484497, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7542981654405594, "clip_ratio/low_mean": 0.03578966949135065, "clip_ratio/low_min": 0.03578966949135065, "clip_ratio/high_mean": 0.08373014815151691, "clip_ratio/high_max": 0.08373014815151691, "clip_ratio/region_mean": 0.11951981764286757, "reward_total_mean": 0.6181088089942932, "reward_meter_mean": 0.7248070240020752, "reward_meter_std": 0.3550230860710144, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8007067441940308, "reward_repeat_soft_std": 0.10040460526943207, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.12631450593471527, "reward_total_composite_mean": 0.6181088089942932, "reward_total_composite_std": 0.17873498797416687} {"timestamp_utc": "2026-04-13T01:03:59Z", "mode": "train", "global_step": 1113, "epoch": 0.11180311401305876, "loss": 0.0362, "grad_norm": 12.487972259521484, "learning_rate": 6.630303030303031e-06, "num_tokens": 2002329.0, "completions/mean_length": 60.875, "completions/min_length": 43.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.8940752744674683, "rewards/meter/std": 0.23851589858531952, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8933209180831909, "rewards/repeat_soft/std": 0.07743450999259949, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.13695022463798523, "rewards/total_composite/mean": 0.7537909746170044, "rewards/total_composite/std": 0.12410390377044678, "reward": 0.7537909746170044, "reward_std": 0.12410389631986618, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1420823037624359, "sampling/sampling_logp_difference/max": 1.363016128540039, "sampling/importance_sampling_ratio/min": 0.25588780641555786, "sampling/importance_sampling_ratio/mean": 1.0274837017059326, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2157562151551247, "clip_ratio/low_mean": 0.007352941203862429, "clip_ratio/low_min": 0.007352941203862429, "clip_ratio/high_mean": 0.13031613919883966, "clip_ratio/high_max": 0.13031613919883966, "clip_ratio/region_mean": 0.1376690804027021, "reward_total_mean": 0.7537909746170044, "reward_meter_mean": 0.8940752744674683, "reward_meter_std": 0.23851589858531952, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8933209180831909, "reward_repeat_soft_std": 0.07743450999259949, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.13695022463798523, "reward_total_composite_mean": 0.7537909746170044, "reward_total_composite_std": 0.12410390377044678} {"timestamp_utc": "2026-04-13T01:04:07Z", "mode": "train", "global_step": 1114, "epoch": 0.11190356604721245, "loss": -0.0437, "grad_norm": 7.604320049285889, "learning_rate": 6.627272727272728e-06, "num_tokens": 2004712.0, "completions/mean_length": 100.875, "completions/min_length": 90.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.875, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.9920145273208618, "rewards/meter/std": 0.005761073436588049, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8701635599136353, "rewards/repeat_soft/std": 0.11659270524978638, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7527979016304016, "rewards/total_composite/std": 0.032765135169029236, "reward": 0.7527979016304016, "reward_std": 0.032765138894319534, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12402273714542389, "sampling/sampling_logp_difference/max": 1.3947834968566895, "sampling/importance_sampling_ratio/min": 0.24788670241832733, "sampling/importance_sampling_ratio/mean": 1.0149492025375366, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9243152588605881, "clip_ratio/low_mean": 0.041738033294677734, "clip_ratio/low_min": 0.041738033294677734, "clip_ratio/high_mean": 0.06147304316982627, "clip_ratio/high_max": 0.06147304316982627, "clip_ratio/region_mean": 0.103211076464504, "reward_total_mean": 0.7527979016304016, "reward_meter_mean": 0.9920145273208618, "reward_meter_std": 0.005761073436588049, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8701635599136353, "reward_repeat_soft_std": 0.11659270524978638, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7527979016304016, "reward_total_composite_std": 0.032765135169029236} {"timestamp_utc": "2026-04-13T01:04:14Z", "mode": "train", "global_step": 1115, "epoch": 0.11200401808136615, "loss": 0.0226, "grad_norm": 11.518064498901367, "learning_rate": 6.624242424242425e-06, "num_tokens": 2006782.0, "completions/mean_length": 72.75, "completions/min_length": 69.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.75, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9909715056419373, "rewards/meter/std": 0.0027424104046076536, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9750661849975586, "rewards/repeat_soft/std": 0.01386494841426611, "rewards/judge_quality/mean": 0.5862499475479126, "rewards/judge_quality/std": 0.17311744391918182, "rewards/total_composite/mean": 0.8318188190460205, "rewards/total_composite/std": 0.051549628376960754, "reward": 0.8318188190460205, "reward_std": 0.05154963582754135, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13052484393119812, "sampling/sampling_logp_difference/max": 1.7207365036010742, "sampling/importance_sampling_ratio/min": 0.17893432080745697, "sampling/importance_sampling_ratio/mean": 1.0258474349975586, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9200575426220894, "clip_ratio/low_mean": 0.05541684478521347, "clip_ratio/low_min": 0.05541684478521347, "clip_ratio/high_mean": 0.06010986864566803, "clip_ratio/high_max": 0.06010986864566803, "clip_ratio/region_mean": 0.1155267134308815, "reward_total_mean": 0.8318188190460205, "reward_meter_mean": 0.9909715056419373, "reward_meter_std": 0.0027424104046076536, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9750661849975586, "reward_repeat_soft_std": 0.01386494841426611, "reward_judge_quality_mean": 0.5862499475479126, "reward_judge_quality_std": 0.17311744391918182, "reward_total_composite_mean": 0.8318188190460205, "reward_total_composite_std": 0.051549628376960754} {"timestamp_utc": "2026-04-13T01:04:21Z", "mode": "train", "global_step": 1116, "epoch": 0.11210447011551984, "loss": 0.0167, "grad_norm": 22.327404022216797, "learning_rate": 6.621212121212121e-06, "num_tokens": 2008351.0, "completions/mean_length": 38.125, "completions/min_length": 31.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.125, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.6370000839233398, "rewards/meter/std": 0.41976141929626465, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9979944229125977, "rewards/repeat_soft/std": 0.002947787521407008, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.6611995100975037, "rewards/total_composite/std": 0.1952054500579834, "reward": 0.6611995100975037, "reward_std": 0.1952054351568222, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1478780061006546, "sampling/sampling_logp_difference/max": 1.627760887145996, "sampling/importance_sampling_ratio/min": 0.19636878371238708, "sampling/importance_sampling_ratio/mean": 1.0075050592422485, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9389638975262642, "clip_ratio/low_mean": 0.03978632064536214, "clip_ratio/low_min": 0.03978632064536214, "clip_ratio/high_mean": 0.10315321013331413, "clip_ratio/high_max": 0.10315321013331413, "clip_ratio/region_mean": 0.14293953077867627, "reward_total_mean": 0.6611995100975037, "reward_meter_mean": 0.6370000839233398, "reward_meter_std": 0.41976141929626465, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9979944229125977, "reward_repeat_soft_std": 0.002947787521407008, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.6611995100975037, "reward_total_composite_std": 0.1952054500579834} {"timestamp_utc": "2026-04-13T01:04:28Z", "mode": "train", "global_step": 1117, "epoch": 0.11220492214967354, "loss": 0.0578, "grad_norm": 16.08549690246582, "learning_rate": 6.618181818181819e-06, "num_tokens": 2010084.0, "completions/mean_length": 43.625, "completions/min_length": 38.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.625, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.34955087304115295, "rewards/meter/std": 0.4029387831687927, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9462816715240479, "rewards/repeat_soft/std": 0.04535457119345665, "rewards/judge_quality/mean": 0.6812499761581421, "rewards/judge_quality/std": 0.25542333722114563, "rewards/total_composite/mean": 0.6063010096549988, "rewards/total_composite/std": 0.23992782831192017, "reward": 0.6063010096549988, "reward_std": 0.23992782831192017, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16894112527370453, "sampling/sampling_logp_difference/max": 2.6417698860168457, "sampling/importance_sampling_ratio/min": 0.07123508304357529, "sampling/importance_sampling_ratio/mean": 0.9980078935623169, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7778629213571548, "clip_ratio/low_mean": 0.057453845627605915, "clip_ratio/low_min": 0.057453845627605915, "clip_ratio/high_mean": 0.08164582587778568, "clip_ratio/high_max": 0.08164582587778568, "clip_ratio/region_mean": 0.1390996715053916, "reward_total_mean": 0.6063010096549988, "reward_meter_mean": 0.34955087304115295, "reward_meter_std": 0.4029387831687927, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9462816715240479, "reward_repeat_soft_std": 0.04535457119345665, "reward_judge_quality_mean": 0.6812499761581421, "reward_judge_quality_std": 0.25542333722114563, "reward_total_composite_mean": 0.6063010096549988, "reward_total_composite_std": 0.23992782831192017} {"timestamp_utc": "2026-04-13T01:04:37Z", "mode": "train", "global_step": 1118, "epoch": 0.11230537418382722, "loss": -0.0028, "grad_norm": 9.003766059875488, "learning_rate": 6.615151515151516e-06, "num_tokens": 2012469.0, "completions/mean_length": 85.125, "completions/min_length": 73.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.125, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.9762908816337585, "rewards/meter/std": 0.045357320457696915, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8682522177696228, "rewards/repeat_soft/std": 0.07767175137996674, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.783406138420105, "rewards/total_composite/std": 0.040346335619688034, "reward": 0.783406138420105, "reward_std": 0.04034634307026863, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1227283775806427, "sampling/sampling_logp_difference/max": 2.392144203186035, "sampling/importance_sampling_ratio/min": 0.09143342077732086, "sampling/importance_sampling_ratio/mean": 1.0053622722625732, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7747612223029137, "clip_ratio/low_mean": 0.0820166738703847, "clip_ratio/low_min": 0.0820166738703847, "clip_ratio/high_mean": 0.03237142227590084, "clip_ratio/high_max": 0.03237142227590084, "clip_ratio/region_mean": 0.11438809614628553, "reward_total_mean": 0.783406138420105, "reward_meter_mean": 0.9762908816337585, "reward_meter_std": 0.045357320457696915, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8682522177696228, "reward_repeat_soft_std": 0.07767175137996674, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.783406138420105, "reward_total_composite_std": 0.040346335619688034} {"timestamp_utc": "2026-04-13T01:04:49Z", "mode": "train", "global_step": 1119, "epoch": 0.11240582621798091, "loss": -0.1754, "grad_norm": 2.0556704998016357, "learning_rate": 6.612121212121213e-06, "num_tokens": 2014313.0, "completions/mean_length": 136.5, "completions/min_length": 70.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 82.85714721679688, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.8685644865036011, "rewards/meter/std": 0.3501354455947876, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9486191272735596, "rewards/repeat_soft/std": 0.05294287949800491, "rewards/judge_quality/mean": 0.5362499952316284, "rewards/judge_quality/std": 0.29731839895248413, "rewards/total_composite/mean": 0.7310037016868591, "rewards/total_composite/std": 0.3025646507740021, "reward": 0.7310037016868591, "reward_std": 0.3025646209716797, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1301787942647934, "sampling/sampling_logp_difference/max": 1.594879150390625, "sampling/importance_sampling_ratio/min": 0.20293305814266205, "sampling/importance_sampling_ratio/mean": 1.0200221538543701, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.038646087050438, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.13654045946896076, "clip_ratio/high_max": 0.13654045946896076, "clip_ratio/region_mean": 0.13654045946896076, "reward_total_mean": 0.7310037016868591, "reward_meter_mean": 0.8685644865036011, "reward_meter_std": 0.3501354455947876, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9486191272735596, "reward_repeat_soft_std": 0.05294287949800491, "reward_judge_quality_mean": 0.5362499952316284, "reward_judge_quality_std": 0.29731839895248413, "reward_total_composite_mean": 0.7310037016868591, "reward_total_composite_std": 0.3025646507740021} {"timestamp_utc": "2026-04-13T01:04:59Z", "mode": "train", "global_step": 1120, "epoch": 0.1125062782521346, "loss": 0.0803, "grad_norm": 9.408740997314453, "learning_rate": 6.609090909090909e-06, "num_tokens": 2016549.0, "completions/mean_length": 99.5, "completions/min_length": 90.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.5, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.4099673628807068, "rewards/meter/std": 0.33178290724754333, "rewards/count_adherence/mean": 0.7749999761581421, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.954979658126831, "rewards/repeat_soft/std": 0.022488312795758247, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.5094833374023438, "rewards/total_composite/std": 0.16481100022792816, "reward": 0.5094833374023438, "reward_std": 0.16481100022792816, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1655120849609375, "sampling/sampling_logp_difference/max": 5.452312469482422, "sampling/importance_sampling_ratio/min": 0.00428638095036149, "sampling/importance_sampling_ratio/mean": 1.0100473165512085, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2972284480929375, "clip_ratio/low_mean": 0.05187926860526204, "clip_ratio/low_min": 0.05187926860526204, "clip_ratio/high_mean": 0.0808100551366806, "clip_ratio/high_max": 0.0808100551366806, "clip_ratio/region_mean": 0.13268932374194264, "reward_total_mean": 0.5094833374023438, "reward_meter_mean": 0.4099673628807068, "reward_meter_std": 0.33178290724754333, "reward_count_adherence_mean": 0.7749999761581421, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.954979658126831, "reward_repeat_soft_std": 0.022488312795758247, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.5094833374023438, "reward_total_composite_std": 0.16481100022792816} {"timestamp_utc": "2026-04-13T01:05:06Z", "mode": "train", "global_step": 1121, "epoch": 0.1126067302862883, "loss": 0.0473, "grad_norm": 9.304445266723633, "learning_rate": 6.606060606060607e-06, "num_tokens": 2018650.0, "completions/mean_length": 83.625, "completions/min_length": 74.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.625, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.6751064658164978, "rewards/meter/std": 0.35764479637145996, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9066567420959473, "rewards/repeat_soft/std": 0.052811067551374435, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.6329635381698608, "rewards/total_composite/std": 0.1576828807592392, "reward": 0.6329635381698608, "reward_std": 0.157682865858078, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13320159912109375, "sampling/sampling_logp_difference/max": 1.5815844535827637, "sampling/importance_sampling_ratio/min": 0.2056489884853363, "sampling/importance_sampling_ratio/mean": 1.0198371410369873, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7454377114772797, "clip_ratio/low_mean": 0.07656387239694595, "clip_ratio/low_min": 0.07656387239694595, "clip_ratio/high_mean": 0.04622564138844609, "clip_ratio/high_max": 0.04622564138844609, "clip_ratio/region_mean": 0.12278951378539205, "reward_total_mean": 0.6329635381698608, "reward_meter_mean": 0.6751064658164978, "reward_meter_std": 0.35764479637145996, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9066567420959473, "reward_repeat_soft_std": 0.052811067551374435, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.6329635381698608, "reward_total_composite_std": 0.1576828807592392} {"timestamp_utc": "2026-04-13T01:05:14Z", "mode": "train", "global_step": 1122, "epoch": 0.112707182320442, "loss": 0.0919, "grad_norm": 9.235660552978516, "learning_rate": 6.603030303030303e-06, "num_tokens": 2020789.0, "completions/mean_length": 82.375, "completions/min_length": 66.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.375, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.7994135022163391, "rewards/meter/std": 0.28798311948776245, "rewards/count_adherence/mean": 0.71875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9197001457214355, "rewards/repeat_soft/std": 0.08854566514492035, "rewards/judge_quality/mean": 0.2887499928474426, "rewards/judge_quality/std": 0.11630470305681229, "rewards/total_composite/mean": 0.6461436152458191, "rewards/total_composite/std": 0.1437547504901886, "reward": 0.6461436152458191, "reward_std": 0.1437547653913498, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13518507778644562, "sampling/sampling_logp_difference/max": 1.61790132522583, "sampling/importance_sampling_ratio/min": 0.19831445813179016, "sampling/importance_sampling_ratio/mean": 1.0251495838165283, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0694303065538406, "clip_ratio/low_mean": 0.04207835718989372, "clip_ratio/low_min": 0.04207835718989372, "clip_ratio/high_mean": 0.07685491722077131, "clip_ratio/high_max": 0.07685491722077131, "clip_ratio/region_mean": 0.11893327441066504, "reward_total_mean": 0.6461436152458191, "reward_meter_mean": 0.7994135022163391, "reward_meter_std": 0.28798311948776245, "reward_count_adherence_mean": 0.71875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9197001457214355, "reward_repeat_soft_std": 0.08854566514492035, "reward_judge_quality_mean": 0.2887499928474426, "reward_judge_quality_std": 0.11630470305681229, "reward_total_composite_mean": 0.6461436152458191, "reward_total_composite_std": 0.1437547504901886} {"timestamp_utc": "2026-04-13T01:05:26Z", "mode": "train", "global_step": 1123, "epoch": 0.11280763435459568, "loss": -0.143, "grad_norm": 2.435908079147339, "learning_rate": 6.600000000000001e-06, "num_tokens": 2022521.0, "completions/mean_length": 113.5, "completions/min_length": 48.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 56.57143020629883, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.7889596819877625, "rewards/meter/std": 0.3347212076187134, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9184024333953857, "rewards/repeat_soft/std": 0.082645945250988, "rewards/judge_quality/mean": 0.26375001668930054, "rewards/judge_quality/std": 0.14401760697364807, "rewards/total_composite/mean": 0.5987436771392822, "rewards/total_composite/std": 0.2484746277332306, "reward": 0.5987436771392822, "reward_std": 0.2484746128320694, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1368652731180191, "sampling/sampling_logp_difference/max": 1.7250447273254395, "sampling/importance_sampling_ratio/min": 0.17816509306430817, "sampling/importance_sampling_ratio/mean": 1.039139747619629, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9993179216980934, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.1110114767216146, "clip_ratio/high_max": 0.1110114767216146, "clip_ratio/region_mean": 0.1110114767216146, "reward_total_mean": 0.5987436771392822, "reward_meter_mean": 0.7889596819877625, "reward_meter_std": 0.3347212076187134, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9184024333953857, "reward_repeat_soft_std": 0.082645945250988, "reward_judge_quality_mean": 0.26375001668930054, "reward_judge_quality_std": 0.14401760697364807, "reward_total_composite_mean": 0.5987436771392822, "reward_total_composite_std": 0.2484746277332306} {"timestamp_utc": "2026-04-13T01:05:37Z", "mode": "train", "global_step": 1124, "epoch": 0.11290808638874937, "loss": -0.1162, "grad_norm": 4.716751575469971, "learning_rate": 6.596969696969698e-06, "num_tokens": 2024626.0, "completions/mean_length": 146.125, "completions/min_length": 84.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 93.85714721679688, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.12131191790103912, "rewards/meter/std": 0.11025381088256836, "rewards/count_adherence/mean": 0.7708333134651184, "rewards/count_adherence/std": 0.1767766773700714, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.8940798044204712, "rewards/repeat_soft/std": 0.09831305593252182, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.13715866208076477, "rewards/total_composite/mean": 0.2879733443260193, "rewards/total_composite/std": 0.18543508648872375, "reward": 0.2879733443260193, "reward_std": 0.18543508648872375, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1340010166168213, "sampling/sampling_logp_difference/max": 2.505736827850342, "sampling/importance_sampling_ratio/min": 0.08161543309688568, "sampling/importance_sampling_ratio/mean": 1.0146633386611938, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.679533377289772, "clip_ratio/low_mean": 0.01308139506727457, "clip_ratio/low_min": 0.01308139506727457, "clip_ratio/high_mean": 0.0940040173009038, "clip_ratio/high_max": 0.0940040173009038, "clip_ratio/region_mean": 0.10708541236817837, "reward_total_mean": 0.2879733443260193, "reward_meter_mean": 0.12131191790103912, "reward_meter_std": 0.11025381088256836, "reward_count_adherence_mean": 0.7708333134651184, "reward_count_adherence_std": 0.1767766773700714, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.8940798044204712, "reward_repeat_soft_std": 0.09831305593252182, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.13715866208076477, "reward_total_composite_mean": 0.2879733443260193, "reward_total_composite_std": 0.18543508648872375} {"timestamp_utc": "2026-04-13T01:05:49Z", "mode": "train", "global_step": 1125, "epoch": 0.11300853842290307, "loss": -0.0936, "grad_norm": 2.06064510345459, "learning_rate": 6.593939393939395e-06, "num_tokens": 2026198.0, "completions/mean_length": 159.5, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 42.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.34633150696754456, "rewards/meter/std": 0.39209237694740295, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9622589349746704, "rewards/repeat_soft/std": 0.017836203798651695, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.1738995909690857, "rewards/total_composite/mean": 0.4322460889816284, "rewards/total_composite/std": 0.306905061006546, "reward": 0.4322460889816284, "reward_std": 0.306905061006546, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19377344846725464, "sampling/sampling_logp_difference/max": 3.5261735916137695, "sampling/importance_sampling_ratio/min": 0.029417265206575394, "sampling/importance_sampling_ratio/mean": 0.9940386414527893, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9721307009458542, "clip_ratio/low_mean": 0.0627507995814085, "clip_ratio/low_min": 0.0627507995814085, "clip_ratio/high_mean": 0.060085996985435486, "clip_ratio/high_max": 0.060085996985435486, "clip_ratio/region_mean": 0.12283679656684399, "reward_total_mean": 0.4322460889816284, "reward_meter_mean": 0.34633150696754456, "reward_meter_std": 0.39209237694740295, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9622589349746704, "reward_repeat_soft_std": 0.017836203798651695, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.1738995909690857, "reward_total_composite_mean": 0.4322460889816284, "reward_total_composite_std": 0.306905061006546} {"timestamp_utc": "2026-04-13T01:05:55Z", "mode": "train", "global_step": 1126, "epoch": 0.11310899045705676, "loss": 0.0532, "grad_norm": 11.183996200561523, "learning_rate": 6.590909090909091e-06, "num_tokens": 2028078.0, "completions/mean_length": 55.0, "completions/min_length": 46.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.0, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.8881584405899048, "rewards/meter/std": 0.2867814600467682, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8110510110855103, "rewards/repeat_soft/std": 0.09295593947172165, "rewards/judge_quality/mean": 0.6525000333786011, "rewards/judge_quality/std": 0.2921227812767029, "rewards/total_composite/mean": 0.826526403427124, "rewards/total_composite/std": 0.17864440381526947, "reward": 0.826526403427124, "reward_std": 0.17864440381526947, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10943987965583801, "sampling/sampling_logp_difference/max": 1.9042892456054688, "sampling/importance_sampling_ratio/min": 0.1489284485578537, "sampling/importance_sampling_ratio/mean": 1.0086448192596436, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7071638628840446, "clip_ratio/low_mean": 0.039467031601816416, "clip_ratio/low_min": 0.039467031601816416, "clip_ratio/high_mean": 0.05625658668577671, "clip_ratio/high_max": 0.05625658668577671, "clip_ratio/region_mean": 0.09572361828759313, "reward_total_mean": 0.826526403427124, "reward_meter_mean": 0.8881584405899048, "reward_meter_std": 0.2867814600467682, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8110510110855103, "reward_repeat_soft_std": 0.09295593947172165, "reward_judge_quality_mean": 0.6525000333786011, "reward_judge_quality_std": 0.2921227812767029, "reward_total_composite_mean": 0.826526403427124, "reward_total_composite_std": 0.17864440381526947} {"timestamp_utc": "2026-04-13T01:06:02Z", "mode": "train", "global_step": 1127, "epoch": 0.11320944249121044, "loss": 0.0574, "grad_norm": 18.137451171875, "learning_rate": 6.5878787878787885e-06, "num_tokens": 2029521.0, "completions/mean_length": 24.375, "completions/min_length": 21.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.375, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.6327652931213379, "rewards/meter/std": 0.47937217354774475, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.956274688243866, "rewards/repeat_soft/std": 0.012105435132980347, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.6796218752861023, "rewards/total_composite/std": 0.23544594645500183, "reward": 0.6796218752861023, "reward_std": 0.23544593155384064, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1783047467470169, "sampling/sampling_logp_difference/max": 1.40631902217865, "sampling/importance_sampling_ratio/min": 0.24504362046718597, "sampling/importance_sampling_ratio/mean": 1.0237740278244019, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2356731370091438, "clip_ratio/low_mean": 0.0406746044754982, "clip_ratio/low_min": 0.0406746044754982, "clip_ratio/high_mean": 0.08499094052240252, "clip_ratio/high_max": 0.08499094052240252, "clip_ratio/region_mean": 0.12566554499790072, "reward_total_mean": 0.6796218752861023, "reward_meter_mean": 0.6327652931213379, "reward_meter_std": 0.47937217354774475, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.956274688243866, "reward_repeat_soft_std": 0.012105435132980347, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.6796218752861023, "reward_total_composite_std": 0.23544594645500183} {"timestamp_utc": "2026-04-13T01:06:08Z", "mode": "train", "global_step": 1128, "epoch": 0.11330989452536414, "loss": 0.1476, "grad_norm": 24.537416458129883, "learning_rate": 6.584848484848485e-06, "num_tokens": 2030988.0, "completions/mean_length": 29.375, "completions/min_length": 25.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.375, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.6674962043762207, "rewards/meter/std": 0.44675543904304504, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9927893877029419, "rewards/repeat_soft/std": 0.013469607569277287, "rewards/judge_quality/mean": 0.7612500190734863, "rewards/judge_quality/std": 0.3026755750179291, "rewards/total_composite/mean": 0.7780272960662842, "rewards/total_composite/std": 0.2249680459499359, "reward": 0.7780272960662842, "reward_std": 0.2249680459499359, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17993266880512238, "sampling/sampling_logp_difference/max": 3.296867847442627, "sampling/importance_sampling_ratio/min": 0.036998871713876724, "sampling/importance_sampling_ratio/mean": 1.0064717531204224, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7581883631646633, "clip_ratio/low_mean": 0.07479587197303772, "clip_ratio/low_min": 0.07479587197303772, "clip_ratio/high_mean": 0.07063492201268673, "clip_ratio/high_max": 0.07063492201268673, "clip_ratio/region_mean": 0.14543079398572445, "reward_total_mean": 0.7780272960662842, "reward_meter_mean": 0.6674962043762207, "reward_meter_std": 0.44675543904304504, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9927893877029419, "reward_repeat_soft_std": 0.013469607569277287, "reward_judge_quality_mean": 0.7612500190734863, "reward_judge_quality_std": 0.3026755750179291, "reward_total_composite_mean": 0.7780272960662842, "reward_total_composite_std": 0.2249680459499359} {"timestamp_utc": "2026-04-13T01:06:20Z", "mode": "train", "global_step": 1129, "epoch": 0.11341034655951783, "loss": -0.1295, "grad_norm": 3.5082955360412598, "learning_rate": 6.581818181818182e-06, "num_tokens": 2032570.0, "completions/mean_length": 109.75, "completions/min_length": 44.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 52.28571701049805, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.7751498818397522, "rewards/meter/std": 0.337580144405365, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9099518656730652, "rewards/repeat_soft/std": 0.09434471279382706, "rewards/judge_quality/mean": 0.32625001668930054, "rewards/judge_quality/std": 0.15592464804649353, "rewards/total_composite/mean": 0.6783126592636108, "rewards/total_composite/std": 0.21428096294403076, "reward": 0.6783126592636108, "reward_std": 0.21428093314170837, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14740656316280365, "sampling/sampling_logp_difference/max": 1.930063247680664, "sampling/importance_sampling_ratio/min": 0.14513902366161346, "sampling/importance_sampling_ratio/mean": 1.0269566774368286, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8281521499156952, "clip_ratio/low_mean": 0.014999999664723873, "clip_ratio/low_min": 0.014999999664723873, "clip_ratio/high_mean": 0.11050077062100172, "clip_ratio/high_max": 0.11050077062100172, "clip_ratio/region_mean": 0.1255007702857256, "reward_total_mean": 0.6783126592636108, "reward_meter_mean": 0.7751498818397522, "reward_meter_std": 0.337580144405365, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9099518656730652, "reward_repeat_soft_std": 0.09434471279382706, "reward_judge_quality_mean": 0.32625001668930054, "reward_judge_quality_std": 0.15592464804649353, "reward_total_composite_mean": 0.6783126592636108, "reward_total_composite_std": 0.21428096294403076} {"timestamp_utc": "2026-04-13T01:06:26Z", "mode": "train", "global_step": 1130, "epoch": 0.11351079859367152, "loss": 0.0096, "grad_norm": 19.33854866027832, "learning_rate": 6.578787878787879e-06, "num_tokens": 2034061.0, "completions/mean_length": 34.375, "completions/min_length": 25.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.375, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.28641971945762634, "rewards/meter/std": 0.34411606192588806, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9585216641426086, "rewards/repeat_soft/std": 0.03845478221774101, "rewards/judge_quality/mean": 0.7074999809265137, "rewards/judge_quality/std": 0.247487410902977, "rewards/total_composite/mean": 0.5369910597801208, "rewards/total_composite/std": 0.177682027220726, "reward": 0.5369910597801208, "reward_std": 0.177682027220726, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12966503202915192, "sampling/sampling_logp_difference/max": 1.3086614608764648, "sampling/importance_sampling_ratio/min": 0.29235681891441345, "sampling/importance_sampling_ratio/mean": 1.03669273853302, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.860918402671814, "clip_ratio/low_mean": 0.06726751662790775, "clip_ratio/low_min": 0.06726751662790775, "clip_ratio/high_mean": 0.05472168745473027, "clip_ratio/high_max": 0.05472168745473027, "clip_ratio/region_mean": 0.12198920408263803, "reward_total_mean": 0.5369910597801208, "reward_meter_mean": 0.28641971945762634, "reward_meter_std": 0.34411606192588806, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9585216641426086, "reward_repeat_soft_std": 0.03845478221774101, "reward_judge_quality_mean": 0.7074999809265137, "reward_judge_quality_std": 0.247487410902977, "reward_total_composite_mean": 0.5369910597801208, "reward_total_composite_std": 0.177682027220726} {"timestamp_utc": "2026-04-13T01:06:32Z", "mode": "train", "global_step": 1131, "epoch": 0.11361125062782522, "loss": 0.0189, "grad_norm": 23.524389266967773, "learning_rate": 6.575757575757577e-06, "num_tokens": 2035814.0, "completions/mean_length": 53.125, "completions/min_length": 45.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.125, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.6507136225700378, "rewards/meter/std": 0.36333873867988586, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9876819252967834, "rewards/repeat_soft/std": 0.008192675188183784, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.6424643397331238, "rewards/total_composite/std": 0.17411908507347107, "reward": 0.6424643397331238, "reward_std": 0.17411908507347107, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15824753046035767, "sampling/sampling_logp_difference/max": 1.9637389183044434, "sampling/importance_sampling_ratio/min": 0.14033274352550507, "sampling/importance_sampling_ratio/mean": 0.999620258808136, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9234117195010185, "clip_ratio/low_mean": 0.0826779268682003, "clip_ratio/low_min": 0.0826779268682003, "clip_ratio/high_mean": 0.09837766364216805, "clip_ratio/high_max": 0.09837766364216805, "clip_ratio/region_mean": 0.18105559051036835, "reward_total_mean": 0.6424643397331238, "reward_meter_mean": 0.6507136225700378, "reward_meter_std": 0.36333873867988586, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9876819252967834, "reward_repeat_soft_std": 0.008192675188183784, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.6424643397331238, "reward_total_composite_std": 0.17411908507347107} {"timestamp_utc": "2026-04-13T01:06:44Z", "mode": "train", "global_step": 1132, "epoch": 0.1137117026619789, "loss": 0.0018, "grad_norm": 2.4035966396331787, "learning_rate": 6.572727272727273e-06, "num_tokens": 2037508.0, "completions/mean_length": 180.75, "completions/min_length": 31.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 70.33333587646484, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 213.0, "rewards/meter/mean": 0.4774877429008484, "rewards/meter/std": 0.44795697927474976, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.25877460837364197, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9780073165893555, "rewards/repeat_soft/std": 0.016996584832668304, "rewards/judge_quality/mean": 0.2574999928474426, "rewards/judge_quality/std": 0.19594095647335052, "rewards/total_composite/mean": 0.37986481189727783, "rewards/total_composite/std": 0.3190499246120453, "reward": 0.37986481189727783, "reward_std": 0.3190499246120453, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1746421605348587, "sampling/sampling_logp_difference/max": 2.659182548522949, "sampling/importance_sampling_ratio/min": 0.07000542432069778, "sampling/importance_sampling_ratio/mean": 1.005347728729248, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8695912659168243, "clip_ratio/low_mean": 0.0035211266949772835, "clip_ratio/low_min": 0.0035211266949772835, "clip_ratio/high_mean": 0.09862828440964222, "clip_ratio/high_max": 0.09862828440964222, "clip_ratio/region_mean": 0.1021494111046195, "reward_total_mean": 0.37986481189727783, "reward_meter_mean": 0.4774877429008484, "reward_meter_std": 0.44795697927474976, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.25877460837364197, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9780073165893555, "reward_repeat_soft_std": 0.016996584832668304, "reward_judge_quality_mean": 0.2574999928474426, "reward_judge_quality_std": 0.19594095647335052, "reward_total_composite_mean": 0.37986481189727783, "reward_total_composite_std": 0.3190499246120453} {"timestamp_utc": "2026-04-13T01:06:56Z", "mode": "train", "global_step": 1133, "epoch": 0.1138121546961326, "loss": -0.0934, "grad_norm": 3.9925436973571777, "learning_rate": 6.56969696969697e-06, "num_tokens": 2039169.0, "completions/mean_length": 101.625, "completions/min_length": 36.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 43.000003814697266, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.47391730546951294, "rewards/meter/std": 0.4226733148097992, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9722276926040649, "rewards/repeat_soft/std": 0.022812888026237488, "rewards/judge_quality/mean": 0.6274999976158142, "rewards/judge_quality/std": 0.3366537094116211, "rewards/total_composite/mean": 0.6154122948646545, "rewards/total_composite/std": 0.3238408863544464, "reward": 0.6154122948646545, "reward_std": 0.3238408863544464, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2051321566104889, "sampling/sampling_logp_difference/max": 2.862928867340088, "sampling/importance_sampling_ratio/min": 0.05710127577185631, "sampling/importance_sampling_ratio/mean": 1.0288535356521606, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1340821012854576, "clip_ratio/low_mean": 0.04096669238060713, "clip_ratio/low_min": 0.04096669238060713, "clip_ratio/high_mean": 0.061737977899610996, "clip_ratio/high_max": 0.061737977899610996, "clip_ratio/region_mean": 0.10270467028021812, "reward_total_mean": 0.6154122948646545, "reward_meter_mean": 0.47391730546951294, "reward_meter_std": 0.4226733148097992, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9722276926040649, "reward_repeat_soft_std": 0.022812888026237488, "reward_judge_quality_mean": 0.6274999976158142, "reward_judge_quality_std": 0.3366537094116211, "reward_total_composite_mean": 0.6154122948646545, "reward_total_composite_std": 0.3238408863544464} {"timestamp_utc": "2026-04-13T01:07:02Z", "mode": "train", "global_step": 1134, "epoch": 0.11391260673028629, "loss": 0.0284, "grad_norm": 12.022897720336914, "learning_rate": 6.566666666666667e-06, "num_tokens": 2041114.0, "completions/mean_length": 55.125, "completions/min_length": 47.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.125, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9948791265487671, "rewards/meter/std": 0.004857604391872883, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.963672399520874, "rewards/repeat_soft/std": 0.04199322313070297, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7636878490447998, "rewards/total_composite/std": 0.01685328036546707, "reward": 0.7636878490447998, "reward_std": 0.016853291541337967, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14382895827293396, "sampling/sampling_logp_difference/max": 1.7106304168701172, "sampling/importance_sampling_ratio/min": 0.18075181543827057, "sampling/importance_sampling_ratio/mean": 1.0069305896759033, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0185871049761772, "clip_ratio/low_mean": 0.03292372776195407, "clip_ratio/low_min": 0.03292372776195407, "clip_ratio/high_mean": 0.11808606795966625, "clip_ratio/high_max": 0.11808606795966625, "clip_ratio/region_mean": 0.15100979572162032, "reward_total_mean": 0.7636878490447998, "reward_meter_mean": 0.9948791265487671, "reward_meter_std": 0.004857604391872883, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.963672399520874, "reward_repeat_soft_std": 0.04199322313070297, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7636878490447998, "reward_total_composite_std": 0.01685328036546707} {"timestamp_utc": "2026-04-13T01:07:14Z", "mode": "train", "global_step": 1135, "epoch": 0.11401305876443998, "loss": -0.1377, "grad_norm": 3.455709934234619, "learning_rate": 6.563636363636364e-06, "num_tokens": 2043011.0, "completions/mean_length": 117.125, "completions/min_length": 49.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 60.71428680419922, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.8839701414108276, "rewards/meter/std": 0.21954423189163208, "rewards/count_adherence/mean": 0.53125, "rewards/count_adherence/std": 0.1602174937725067, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9912860989570618, "rewards/repeat_soft/std": 0.008290661498904228, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.13845446705818176, "rewards/total_composite/mean": 0.6155022382736206, "rewards/total_composite/std": 0.2640216648578644, "reward": 0.6155022382736206, "reward_std": 0.2640216648578644, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18092848360538483, "sampling/sampling_logp_difference/max": 2.6546154022216797, "sampling/importance_sampling_ratio/min": 0.07032588124275208, "sampling/importance_sampling_ratio/mean": 1.0225123167037964, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.252980962395668, "clip_ratio/low_mean": 0.013888888992369175, "clip_ratio/low_min": 0.013888888992369175, "clip_ratio/high_mean": 0.12882943637669086, "clip_ratio/high_max": 0.12882943637669086, "clip_ratio/region_mean": 0.14271832536906004, "reward_total_mean": 0.6155022382736206, "reward_meter_mean": 0.8839701414108276, "reward_meter_std": 0.21954423189163208, "reward_count_adherence_mean": 0.53125, "reward_count_adherence_std": 0.1602174937725067, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9912860989570618, "reward_repeat_soft_std": 0.008290661498904228, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.13845446705818176, "reward_total_composite_mean": 0.6155022382736206, "reward_total_composite_std": 0.2640216648578644} {"timestamp_utc": "2026-04-13T01:07:20Z", "mode": "train", "global_step": 1136, "epoch": 0.11411351079859366, "loss": 0.01, "grad_norm": 11.30797290802002, "learning_rate": 6.56060606060606e-06, "num_tokens": 2044719.0, "completions/mean_length": 55.5, "completions/min_length": 52.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.5, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9862388372421265, "rewards/meter/std": 0.020955197513103485, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8597394227981567, "rewards/repeat_soft/std": 0.08657903969287872, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.0975411981344223, "rewards/total_composite/mean": 0.7967814207077026, "rewards/total_composite/std": 0.030098868533968925, "reward": 0.7967814207077026, "reward_std": 0.03009887784719467, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11017373204231262, "sampling/sampling_logp_difference/max": 2.7325057983398438, "sampling/importance_sampling_ratio/min": 0.06505607068538666, "sampling/importance_sampling_ratio/mean": 1.0034085512161255, "sampling/importance_sampling_ratio/max": 1.701078176498413, "entropy": 0.7072731554508209, "clip_ratio/low_mean": 0.018308081198483706, "clip_ratio/low_min": 0.018308081198483706, "clip_ratio/high_mean": 0.0710253156721592, "clip_ratio/high_max": 0.0710253156721592, "clip_ratio/region_mean": 0.0893333968706429, "reward_total_mean": 0.7967814207077026, "reward_meter_mean": 0.9862388372421265, "reward_meter_std": 0.020955197513103485, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8597394227981567, "reward_repeat_soft_std": 0.08657903969287872, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.0975411981344223, "reward_total_composite_mean": 0.7967814207077026, "reward_total_composite_std": 0.030098868533968925} {"timestamp_utc": "2026-04-13T01:07:27Z", "mode": "train", "global_step": 1137, "epoch": 0.11421396283274736, "loss": 0.0377, "grad_norm": 11.82158374786377, "learning_rate": 6.5575757575757585e-06, "num_tokens": 2046311.0, "completions/mean_length": 44.0, "completions/min_length": 39.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.0, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9874078631401062, "rewards/meter/std": 0.008009638637304306, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9071553349494934, "rewards/repeat_soft/std": 0.056527867913246155, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.8507990837097168, "rewards/total_composite/std": 0.06599977612495422, "reward": 0.8507990837097168, "reward_std": 0.06599979102611542, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1141263097524643, "sampling/sampling_logp_difference/max": 1.766200304031372, "sampling/importance_sampling_ratio/min": 0.17098143696784973, "sampling/importance_sampling_ratio/mean": 1.0409996509552002, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.954748347401619, "clip_ratio/low_mean": 0.08003347995691001, "clip_ratio/low_min": 0.08003347995691001, "clip_ratio/high_mean": 0.012500000186264515, "clip_ratio/high_max": 0.012500000186264515, "clip_ratio/region_mean": 0.09253348014317453, "reward_total_mean": 0.8507990837097168, "reward_meter_mean": 0.9874078631401062, "reward_meter_std": 0.008009638637304306, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9071553349494934, "reward_repeat_soft_std": 0.056527867913246155, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.8507990837097168, "reward_total_composite_std": 0.06599977612495422} {"timestamp_utc": "2026-04-13T01:07:34Z", "mode": "train", "global_step": 1138, "epoch": 0.11431441486690105, "loss": 0.0159, "grad_norm": 18.876127243041992, "learning_rate": 6.554545454545455e-06, "num_tokens": 2047843.0, "completions/mean_length": 34.5, "completions/min_length": 30.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.5, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.28148940205574036, "rewards/meter/std": 0.27478229999542236, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9723601937294006, "rewards/repeat_soft/std": 0.012815372087061405, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.5385312438011169, "rewards/total_composite/std": 0.11508198827505112, "reward": 0.5385312438011169, "reward_std": 0.11508196592330933, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16056622564792633, "sampling/sampling_logp_difference/max": 1.877190351486206, "sampling/importance_sampling_ratio/min": 0.15301944315433502, "sampling/importance_sampling_ratio/mean": 1.0010915994644165, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9336200803518295, "clip_ratio/low_mean": 0.08552636788226664, "clip_ratio/low_min": 0.08552636788226664, "clip_ratio/high_mean": 0.05413851421326399, "clip_ratio/high_max": 0.05413851421326399, "clip_ratio/region_mean": 0.13966488209553063, "reward_total_mean": 0.5385312438011169, "reward_meter_mean": 0.28148940205574036, "reward_meter_std": 0.27478229999542236, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9723601937294006, "reward_repeat_soft_std": 0.012815372087061405, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.5385312438011169, "reward_total_composite_std": 0.11508198827505112} {"timestamp_utc": "2026-04-13T01:07:45Z", "mode": "train", "global_step": 1139, "epoch": 0.11441486690105475, "loss": -0.0831, "grad_norm": 2.4249427318573, "learning_rate": 6.551515151515152e-06, "num_tokens": 2049340.0, "completions/mean_length": 157.125, "completions/min_length": 31.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 38.833335876464844, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.4193493127822876, "rewards/meter/std": 0.4269815683364868, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9709213972091675, "rewards/repeat_soft/std": 0.020779425278306007, "rewards/judge_quality/mean": 0.2762500047683716, "rewards/judge_quality/std": 0.17369410395622253, "rewards/total_composite/mean": 0.4496230185031891, "rewards/total_composite/std": 0.3194378614425659, "reward": 0.4496230185031891, "reward_std": 0.3194378614425659, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14678242802619934, "sampling/sampling_logp_difference/max": 1.05880606174469, "sampling/importance_sampling_ratio/min": 0.34686967730522156, "sampling/importance_sampling_ratio/mean": 1.0007442235946655, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8768078237771988, "clip_ratio/low_mean": 0.02297008642926812, "clip_ratio/low_min": 0.02297008642926812, "clip_ratio/high_mean": 0.08378891553729773, "clip_ratio/high_max": 0.08378891553729773, "clip_ratio/region_mean": 0.10675900196656585, "reward_total_mean": 0.4496230185031891, "reward_meter_mean": 0.4193493127822876, "reward_meter_std": 0.4269815683364868, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9709213972091675, "reward_repeat_soft_std": 0.020779425278306007, "reward_judge_quality_mean": 0.2762500047683716, "reward_judge_quality_std": 0.17369410395622253, "reward_total_composite_mean": 0.4496230185031891, "reward_total_composite_std": 0.3194378614425659} {"timestamp_utc": "2026-04-13T01:07:53Z", "mode": "train", "global_step": 1140, "epoch": 0.11451531893520844, "loss": 0.0387, "grad_norm": 12.06127643585205, "learning_rate": 6.5484848484848494e-06, "num_tokens": 2051475.0, "completions/mean_length": 70.875, "completions/min_length": 58.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.875, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.7912387847900391, "rewards/meter/std": 0.24325551092624664, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9533345699310303, "rewards/repeat_soft/std": 0.0534462034702301, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.14520922303199768, "rewards/total_composite/mean": 0.6790158748626709, "rewards/total_composite/std": 0.12709781527519226, "reward": 0.6790158748626709, "reward_std": 0.12709781527519226, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1491783857345581, "sampling/sampling_logp_difference/max": 2.2076730728149414, "sampling/importance_sampling_ratio/min": 0.10995621234178543, "sampling/importance_sampling_ratio/mean": 1.0120470523834229, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0912744104862213, "clip_ratio/low_mean": 0.058624787256121635, "clip_ratio/low_min": 0.058624787256121635, "clip_ratio/high_mean": 0.07741347420960665, "clip_ratio/high_max": 0.07741347420960665, "clip_ratio/region_mean": 0.13603826146572828, "reward_total_mean": 0.6790158748626709, "reward_meter_mean": 0.7912387847900391, "reward_meter_std": 0.24325551092624664, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9533345699310303, "reward_repeat_soft_std": 0.0534462034702301, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.14520922303199768, "reward_total_composite_mean": 0.6790158748626709, "reward_total_composite_std": 0.12709781527519226} {"timestamp_utc": "2026-04-13T01:08:00Z", "mode": "train", "global_step": 1141, "epoch": 0.11461577096936212, "loss": 0.0383, "grad_norm": 12.577277183532715, "learning_rate": 6.545454545454546e-06, "num_tokens": 2053097.0, "completions/mean_length": 48.75, "completions/min_length": 43.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.75, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.6512957811355591, "rewards/meter/std": 0.29529187083244324, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9662147760391235, "rewards/repeat_soft/std": 0.02304602973163128, "rewards/judge_quality/mean": 0.41749998927116394, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.6649546027183533, "rewards/total_composite/std": 0.1510412096977234, "reward": 0.6649546027183533, "reward_std": 0.1510411947965622, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18179255723953247, "sampling/sampling_logp_difference/max": 2.126775026321411, "sampling/importance_sampling_ratio/min": 0.11922115832567215, "sampling/importance_sampling_ratio/mean": 1.0029736757278442, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3659209534525871, "clip_ratio/low_mean": 0.07158080488443375, "clip_ratio/low_min": 0.07158080488443375, "clip_ratio/high_mean": 0.09720815904438496, "clip_ratio/high_max": 0.09720815904438496, "clip_ratio/region_mean": 0.1687889639288187, "reward_total_mean": 0.6649546027183533, "reward_meter_mean": 0.6512957811355591, "reward_meter_std": 0.29529187083244324, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9662147760391235, "reward_repeat_soft_std": 0.02304602973163128, "reward_judge_quality_mean": 0.41749998927116394, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.6649546027183533, "reward_total_composite_std": 0.1510412096977234} {"timestamp_utc": "2026-04-13T01:08:06Z", "mode": "train", "global_step": 1142, "epoch": 0.11471622300351582, "loss": 0.0425, "grad_norm": 13.634198188781738, "learning_rate": 6.542424242424243e-06, "num_tokens": 2054858.0, "completions/mean_length": 58.125, "completions/min_length": 53.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.125, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.2619219720363617, "rewards/meter/std": 0.21396492421627045, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9641907215118408, "rewards/repeat_soft/std": 0.02886543795466423, "rewards/judge_quality/mean": 0.5950000286102295, "rewards/judge_quality/std": 0.19820626080036163, "rewards/total_composite/mean": 0.5052839517593384, "rewards/total_composite/std": 0.12820802628993988, "reward": 0.5052839517593384, "reward_std": 0.12820802628993988, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15375690162181854, "sampling/sampling_logp_difference/max": 2.7344136238098145, "sampling/importance_sampling_ratio/min": 0.06493207067251205, "sampling/importance_sampling_ratio/mean": 1.005129098892212, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7191236019134521, "clip_ratio/low_mean": 0.040108103305101395, "clip_ratio/low_min": 0.040108103305101395, "clip_ratio/high_mean": 0.09713254496455193, "clip_ratio/high_max": 0.09713254496455193, "clip_ratio/region_mean": 0.13724064826965332, "reward_total_mean": 0.5052839517593384, "reward_meter_mean": 0.2619219720363617, "reward_meter_std": 0.21396492421627045, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9641907215118408, "reward_repeat_soft_std": 0.02886543795466423, "reward_judge_quality_mean": 0.5950000286102295, "reward_judge_quality_std": 0.19820626080036163, "reward_total_composite_mean": 0.5052839517593384, "reward_total_composite_std": 0.12820802628993988} {"timestamp_utc": "2026-04-13T01:08:13Z", "mode": "train", "global_step": 1143, "epoch": 0.11481667503766951, "loss": 0.0414, "grad_norm": 36.25640869140625, "learning_rate": 6.5393939393939395e-06, "num_tokens": 2056416.0, "completions/mean_length": 40.75, "completions/min_length": 37.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.75, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.8897489309310913, "rewards/meter/std": 0.2565496563911438, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9316392540931702, "rewards/repeat_soft/std": 0.06010271608829498, "rewards/judge_quality/mean": 0.6525000333786011, "rewards/judge_quality/std": 0.2921227812767029, "rewards/total_composite/mean": 0.7458577752113342, "rewards/total_composite/std": 0.3211231231689453, "reward": 0.7458577752113342, "reward_std": 0.3211230933666229, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16250959038734436, "sampling/sampling_logp_difference/max": 2.110299587249756, "sampling/importance_sampling_ratio/min": 0.1225302666425705, "sampling/importance_sampling_ratio/mean": 1.0165033340454102, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8488689437508583, "clip_ratio/low_mean": 0.030952381435781717, "clip_ratio/low_min": 0.030952381435781717, "clip_ratio/high_mean": 0.107974112033844, "clip_ratio/high_max": 0.107974112033844, "clip_ratio/region_mean": 0.1389264934696257, "reward_total_mean": 0.7458577752113342, "reward_meter_mean": 0.8897489309310913, "reward_meter_std": 0.2565496563911438, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9316392540931702, "reward_repeat_soft_std": 0.06010271608829498, "reward_judge_quality_mean": 0.6525000333786011, "reward_judge_quality_std": 0.2921227812767029, "reward_total_composite_mean": 0.7458577752113342, "reward_total_composite_std": 0.3211231231689453} {"timestamp_utc": "2026-04-13T01:08:20Z", "mode": "train", "global_step": 1144, "epoch": 0.11491712707182321, "loss": 0.0288, "grad_norm": 7.6249308586120605, "learning_rate": 6.536363636363638e-06, "num_tokens": 2058461.0, "completions/mean_length": 82.625, "completions/min_length": 75.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.625, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9760454297065735, "rewards/meter/std": 0.04177071154117584, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7667751908302307, "rewards/repeat_soft/std": 0.12179012596607208, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.12631450593471527, "rewards/total_composite/mean": 0.7277729511260986, "rewards/total_composite/std": 0.04693000018596649, "reward": 0.7277729511260986, "reward_std": 0.046929981559515, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11793001741170883, "sampling/sampling_logp_difference/max": 1.8790535926818848, "sampling/importance_sampling_ratio/min": 0.15273459255695343, "sampling/importance_sampling_ratio/mean": 1.012752652168274, "sampling/importance_sampling_ratio/max": 1.9637659788131714, "entropy": 0.7091062068939209, "clip_ratio/low_mean": 0.03429903741925955, "clip_ratio/low_min": 0.03429903741925955, "clip_ratio/high_mean": 0.05922740511596203, "clip_ratio/high_max": 0.05922740511596203, "clip_ratio/region_mean": 0.09352644253522158, "reward_total_mean": 0.7277729511260986, "reward_meter_mean": 0.9760454297065735, "reward_meter_std": 0.04177071154117584, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7667751908302307, "reward_repeat_soft_std": 0.12179012596607208, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.12631450593471527, "reward_total_composite_mean": 0.7277729511260986, "reward_total_composite_std": 0.04693000018596649} {"timestamp_utc": "2026-04-13T01:08:26Z", "mode": "train", "global_step": 1145, "epoch": 0.1150175791059769, "loss": 0.0379, "grad_norm": 7.336934566497803, "learning_rate": 6.533333333333334e-06, "num_tokens": 2060771.0, "completions/mean_length": 89.75, "completions/min_length": 78.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.75, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.8899445533752441, "rewards/meter/std": 0.2855664789676666, "rewards/count_adherence/mean": 0.6000000238418579, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8575776815414429, "rewards/repeat_soft/std": 0.19266635179519653, "rewards/judge_quality/mean": 0.25, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.6512328386306763, "rewards/total_composite/std": 0.14550259709358215, "reward": 0.6512328386306763, "reward_std": 0.14550258219242096, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1344706118106842, "sampling/sampling_logp_difference/max": 6.736697673797607, "sampling/importance_sampling_ratio/min": 0.0011865590931847692, "sampling/importance_sampling_ratio/mean": 0.9968032240867615, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7238908112049103, "clip_ratio/low_mean": 0.010526316240429878, "clip_ratio/low_min": 0.010526316240429878, "clip_ratio/high_mean": 0.11944214906543493, "clip_ratio/high_max": 0.11944214906543493, "clip_ratio/region_mean": 0.1299684653058648, "reward_total_mean": 0.6512328386306763, "reward_meter_mean": 0.8899445533752441, "reward_meter_std": 0.2855664789676666, "reward_count_adherence_mean": 0.6000000238418579, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8575776815414429, "reward_repeat_soft_std": 0.19266635179519653, "reward_judge_quality_mean": 0.25, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.6512328386306763, "reward_total_composite_std": 0.14550259709358215} {"timestamp_utc": "2026-04-13T01:08:38Z", "mode": "train", "global_step": 1146, "epoch": 0.11511803114013058, "loss": -0.0692, "grad_norm": 4.876411437988281, "learning_rate": 6.530303030303031e-06, "num_tokens": 2062324.0, "completions/mean_length": 103.125, "completions/min_length": 38.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 44.71428680419922, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.5060253143310547, "rewards/meter/std": 0.4789236783981323, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9485106468200684, "rewards/repeat_soft/std": 0.05253385752439499, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.22414520382881165, "rewards/total_composite/mean": 0.5948124527931213, "rewards/total_composite/std": 0.2647015154361725, "reward": 0.5948124527931213, "reward_std": 0.2647015154361725, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15675991773605347, "sampling/sampling_logp_difference/max": 1.966752529144287, "sampling/importance_sampling_ratio/min": 0.18917478621006012, "sampling/importance_sampling_ratio/mean": 1.0167564153671265, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9883990362286568, "clip_ratio/low_mean": 0.04321406967937946, "clip_ratio/low_min": 0.04321406967937946, "clip_ratio/high_mean": 0.06764603685587645, "clip_ratio/high_max": 0.06764603685587645, "clip_ratio/region_mean": 0.11086010653525591, "reward_total_mean": 0.5948124527931213, "reward_meter_mean": 0.5060253143310547, "reward_meter_std": 0.4789236783981323, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9485106468200684, "reward_repeat_soft_std": 0.05253385752439499, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.22414520382881165, "reward_total_composite_mean": 0.5948124527931213, "reward_total_composite_std": 0.2647015154361725} {"timestamp_utc": "2026-04-13T01:08:45Z", "mode": "train", "global_step": 1147, "epoch": 0.11521848317428428, "loss": 0.0575, "grad_norm": 18.816001892089844, "learning_rate": 6.527272727272728e-06, "num_tokens": 2064163.0, "completions/mean_length": 43.875, "completions/min_length": 40.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.875, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.24461744725704193, "rewards/meter/std": 0.33776283264160156, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9926018118858337, "rewards/repeat_soft/std": 0.01157908234745264, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.560338020324707, "rewards/total_composite/std": 0.21185487508773804, "reward": 0.560338020324707, "reward_std": 0.21185486018657684, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18665680289268494, "sampling/sampling_logp_difference/max": 2.993267297744751, "sampling/importance_sampling_ratio/min": 0.05012340471148491, "sampling/importance_sampling_ratio/mean": 1.008411169052124, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0261183381080627, "clip_ratio/low_mean": 0.08174833562225103, "clip_ratio/low_min": 0.08174833562225103, "clip_ratio/high_mean": 0.08071789611130953, "clip_ratio/high_max": 0.08071789611130953, "clip_ratio/region_mean": 0.16246623173356056, "reward_total_mean": 0.560338020324707, "reward_meter_mean": 0.24461744725704193, "reward_meter_std": 0.33776283264160156, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9926018118858337, "reward_repeat_soft_std": 0.01157908234745264, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.560338020324707, "reward_total_composite_std": 0.21185487508773804} {"timestamp_utc": "2026-04-13T01:08:52Z", "mode": "train", "global_step": 1148, "epoch": 0.11531893520843797, "loss": 0.0621, "grad_norm": 14.233384132385254, "learning_rate": 6.524242424242425e-06, "num_tokens": 2065811.0, "completions/mean_length": 51.0, "completions/min_length": 45.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.0, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9460190534591675, "rewards/meter/std": 0.0805860236287117, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9779989719390869, "rewards/repeat_soft/std": 0.009015518240630627, "rewards/judge_quality/mean": 0.5387499928474426, "rewards/judge_quality/std": 0.18192915618419647, "rewards/total_composite/mean": 0.83513343334198, "rewards/total_composite/std": 0.05529789254069328, "reward": 0.83513343334198, "reward_std": 0.05529789626598358, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1578235924243927, "sampling/sampling_logp_difference/max": 1.3650002479553223, "sampling/importance_sampling_ratio/min": 0.2553806006908417, "sampling/importance_sampling_ratio/mean": 1.0058602094650269, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.885134868323803, "clip_ratio/low_mean": 0.1386919841170311, "clip_ratio/low_min": 0.1386919841170311, "clip_ratio/high_mean": 0.02393617108464241, "clip_ratio/high_max": 0.02393617108464241, "clip_ratio/region_mean": 0.1626281552016735, "reward_total_mean": 0.83513343334198, "reward_meter_mean": 0.9460190534591675, "reward_meter_std": 0.0805860236287117, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9779989719390869, "reward_repeat_soft_std": 0.009015518240630627, "reward_judge_quality_mean": 0.5387499928474426, "reward_judge_quality_std": 0.18192915618419647, "reward_total_composite_mean": 0.83513343334198, "reward_total_composite_std": 0.05529789254069328} {"timestamp_utc": "2026-04-13T01:08:58Z", "mode": "train", "global_step": 1149, "epoch": 0.11541938724259167, "loss": 0.0123, "grad_norm": 13.462254524230957, "learning_rate": 6.521212121212121e-06, "num_tokens": 2067710.0, "completions/mean_length": 63.375, "completions/min_length": 51.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.375, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9910237789154053, "rewards/meter/std": 0.006321941036731005, "rewards/count_adherence/mean": 0.5625, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9284415245056152, "rewards/repeat_soft/std": 0.06902176886796951, "rewards/judge_quality/mean": 0.4974999725818634, "rewards/judge_quality/std": 0.2594912648200989, "rewards/total_composite/mean": 0.7724298238754272, "rewards/total_composite/std": 0.08380258083343506, "reward": 0.7724298238754272, "reward_std": 0.08380255848169327, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1216861829161644, "sampling/sampling_logp_difference/max": 2.6602468490600586, "sampling/importance_sampling_ratio/min": 0.06993095576763153, "sampling/importance_sampling_ratio/mean": 0.999319851398468, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6964538283646107, "clip_ratio/low_mean": 0.062073024455457926, "clip_ratio/low_min": 0.062073024455457926, "clip_ratio/high_mean": 0.021915584802627563, "clip_ratio/high_max": 0.021915584802627563, "clip_ratio/region_mean": 0.08398860925808549, "reward_total_mean": 0.7724298238754272, "reward_meter_mean": 0.9910237789154053, "reward_meter_std": 0.006321941036731005, "reward_count_adherence_mean": 0.5625, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9284415245056152, "reward_repeat_soft_std": 0.06902176886796951, "reward_judge_quality_mean": 0.4974999725818634, "reward_judge_quality_std": 0.2594912648200989, "reward_total_composite_mean": 0.7724298238754272, "reward_total_composite_std": 0.08380258083343506} {"timestamp_utc": "2026-04-13T01:09:06Z", "mode": "train", "global_step": 1150, "epoch": 0.11551983927674535, "loss": -0.0353, "grad_norm": 9.55788516998291, "learning_rate": 6.5181818181818195e-06, "num_tokens": 2070106.0, "completions/mean_length": 128.5, "completions/min_length": 106.0, "completions/max_length": 144.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.5, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 144.0, "rewards/meter/mean": 0.603964626789093, "rewards/meter/std": 0.45505228638648987, "rewards/count_adherence/mean": 0.7916666269302368, "rewards/count_adherence/std": 0.07715165615081787, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8657242059707642, "rewards/repeat_soft/std": 0.0974850058555603, "rewards/judge_quality/mean": 0.4737499952316284, "rewards/judge_quality/std": 0.16291432082653046, "rewards/total_composite/mean": 0.6192315220832825, "rewards/total_composite/std": 0.22254277765750885, "reward": 0.6192315220832825, "reward_std": 0.22254276275634766, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14691396057605743, "sampling/sampling_logp_difference/max": 2.1461050510406494, "sampling/importance_sampling_ratio/min": 0.11693874001502991, "sampling/importance_sampling_ratio/mean": 0.999319851398468, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7089584246277809, "clip_ratio/low_mean": 0.04529268480837345, "clip_ratio/low_min": 0.04529268480837345, "clip_ratio/high_mean": 0.08127360697835684, "clip_ratio/high_max": 0.08127360697835684, "clip_ratio/region_mean": 0.1265662917867303, "reward_total_mean": 0.6192315220832825, "reward_meter_mean": 0.603964626789093, "reward_meter_std": 0.45505228638648987, "reward_count_adherence_mean": 0.7916666269302368, "reward_count_adherence_std": 0.07715165615081787, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8657242059707642, "reward_repeat_soft_std": 0.0974850058555603, "reward_judge_quality_mean": 0.4737499952316284, "reward_judge_quality_std": 0.16291432082653046, "reward_total_composite_mean": 0.6192315220832825, "reward_total_composite_std": 0.22254277765750885} {"timestamp_utc": "2026-04-13T01:10:15Z", "mode": "eval", "global_step": 1150, "epoch": 0.11551983927674535, "eval_loss": NaN, "eval_runtime": 69.2669, "eval_samples_per_second": 1.155, "eval_steps_per_second": 0.144, "eval_num_tokens": 2070106.0, "eval_completions/mean_length": 89.0, "eval_completions/min_length": 34.3, "eval_completions/max_length": 309.9, "eval_completions/clipped_ratio": 0.0625, "eval_completions/mean_terminated_length": 60.35357246398926, "eval_completions/min_terminated_length": 34.3, "eval_completions/max_terminated_length": 93.8, "eval_rewards/meter/mean": 0.7352676451206207, "eval_rewards/meter/std": 0.30747348442673683, "eval_rewards/count_adherence/mean": 0.783958351612091, "eval_rewards/count_adherence/std": 0.19223182052373886, "eval_rewards/hard_gate/mean": 0.95, "eval_rewards/hard_gate/std": 0.1414213538169861, "eval_rewards/repeat_soft/mean": 0.9214207053184509, "eval_rewards/repeat_soft/std": 0.06825453042984009, "eval_rewards/judge_quality/mean": 0.39324999749660494, "eval_rewards/judge_quality/std": 0.14997769594192506, "eval_rewards/total_composite/mean": 0.6438218772411346, "eval_rewards/total_composite/std": 0.20464328601956366, "eval_reward": 0.6438218772411346, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.07076848670840263, "eval_sampling/sampling_logp_difference/max": 0.9832262516021728, "eval_sampling/importance_sampling_ratio/min": 0.3862922728061676, "eval_sampling/importance_sampling_ratio/mean": 1.0131242513656615, "eval_sampling/importance_sampling_ratio/max": 1.3783972024917603, "eval_entropy": 0.7613081276416779, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6438218772411346, "eval_reward_meter_mean": 0.7352676451206207, "eval_reward_meter_std": 0.30747348442673683, "eval_reward_count_adherence_mean": 0.783958351612091, "eval_reward_count_adherence_std": 0.19223182052373886, "eval_reward_hard_gate_mean": 0.95, "eval_reward_hard_gate_std": 0.1414213538169861, "eval_reward_repeat_soft_mean": 0.9214207053184509, "eval_reward_repeat_soft_std": 0.06825453042984009, "eval_reward_judge_quality_mean": 0.39324999749660494, "eval_reward_judge_quality_std": 0.14997769594192506, "eval_reward_total_composite_mean": 0.6438218772411346, "eval_reward_total_composite_std": 0.20464328601956366} {"timestamp_utc": "2026-04-13T01:10:25Z", "mode": "train", "global_step": 1151, "epoch": 0.11562029131089904, "loss": 0.0485, "grad_norm": 21.91214942932129, "learning_rate": 6.515151515151516e-06, "num_tokens": 2071664.0, "completions/mean_length": 37.75, "completions/min_length": 35.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.75, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.5580196380615234, "rewards/meter/std": 0.300456702709198, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.992998480796814, "rewards/repeat_soft/std": 0.014231822453439236, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.5974087119102478, "rewards/total_composite/std": 0.11802207678556442, "reward": 0.5974087119102478, "reward_std": 0.11802207678556442, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1643591821193695, "sampling/sampling_logp_difference/max": 1.6648330688476562, "sampling/importance_sampling_ratio/min": 0.18922223150730133, "sampling/importance_sampling_ratio/mean": 0.9971727728843689, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6328787058591843, "clip_ratio/low_mean": 0.08715623617172241, "clip_ratio/low_min": 0.08715623617172241, "clip_ratio/high_mean": 0.08265098836272955, "clip_ratio/high_max": 0.08265098836272955, "clip_ratio/region_mean": 0.16980722453445196, "reward_total_mean": 0.5974087119102478, "reward_meter_mean": 0.5580196380615234, "reward_meter_std": 0.300456702709198, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.992998480796814, "reward_repeat_soft_std": 0.014231822453439236, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.5974087119102478, "reward_total_composite_std": 0.11802207678556442} {"timestamp_utc": "2026-04-13T01:10:32Z", "mode": "train", "global_step": 1152, "epoch": 0.11572074334505274, "loss": 0.0175, "grad_norm": 10.89460563659668, "learning_rate": 6.512121212121213e-06, "num_tokens": 2073268.0, "completions/mean_length": 50.5, "completions/min_length": 46.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.5, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9894185066223145, "rewards/meter/std": 0.008653522469103336, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7409875392913818, "rewards/repeat_soft/std": 0.15474116802215576, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.14520922303199768, "rewards/total_composite/mean": 0.7938370704650879, "rewards/total_composite/std": 0.047543205320835114, "reward": 0.7938370704650879, "reward_std": 0.04754319041967392, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09264032542705536, "sampling/sampling_logp_difference/max": 2.5445384979248047, "sampling/importance_sampling_ratio/min": 0.07850927859544754, "sampling/importance_sampling_ratio/mean": 1.0003619194030762, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44682133570313454, "clip_ratio/low_mean": 0.03498065425083041, "clip_ratio/low_min": 0.03498065425083041, "clip_ratio/high_mean": 0.04101923853158951, "clip_ratio/high_max": 0.04101923853158951, "clip_ratio/region_mean": 0.07599989278241992, "reward_total_mean": 0.7938370704650879, "reward_meter_mean": 0.9894185066223145, "reward_meter_std": 0.008653522469103336, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7409875392913818, "reward_repeat_soft_std": 0.15474116802215576, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.14520922303199768, "reward_total_composite_mean": 0.7938370704650879, "reward_total_composite_std": 0.047543205320835114} {"timestamp_utc": "2026-04-13T01:10:38Z", "mode": "train", "global_step": 1153, "epoch": 0.11582119537920643, "loss": 0.1491, "grad_norm": 17.15336036682129, "learning_rate": 6.5090909090909095e-06, "num_tokens": 2074794.0, "completions/mean_length": 34.75, "completions/min_length": 28.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.75, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.2286786288022995, "rewards/meter/std": 0.2802422344684601, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9574989676475525, "rewards/repeat_soft/std": 0.041296929121017456, "rewards/judge_quality/mean": 0.5724999904632568, "rewards/judge_quality/std": 0.24944224953651428, "rewards/total_composite/mean": 0.4704052805900574, "rewards/total_composite/std": 0.1403525471687317, "reward": 0.4704052805900574, "reward_std": 0.14035256206989288, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13674002885818481, "sampling/sampling_logp_difference/max": 1.167710542678833, "sampling/importance_sampling_ratio/min": 0.3110783100128174, "sampling/importance_sampling_ratio/mean": 0.9957209825515747, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7834102474153042, "clip_ratio/low_mean": 0.05462185014039278, "clip_ratio/low_min": 0.05462185014039278, "clip_ratio/high_mean": 0.10435168631374836, "clip_ratio/high_max": 0.10435168631374836, "clip_ratio/region_mean": 0.15897353645414114, "reward_total_mean": 0.4704052805900574, "reward_meter_mean": 0.2286786288022995, "reward_meter_std": 0.2802422344684601, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9574989676475525, "reward_repeat_soft_std": 0.041296929121017456, "reward_judge_quality_mean": 0.5724999904632568, "reward_judge_quality_std": 0.24944224953651428, "reward_total_composite_mean": 0.4704052805900574, "reward_total_composite_std": 0.1403525471687317} {"timestamp_utc": "2026-04-13T01:10:44Z", "mode": "train", "global_step": 1154, "epoch": 0.11592164741336013, "loss": 0.0218, "grad_norm": 12.609972953796387, "learning_rate": 6.506060606060607e-06, "num_tokens": 2076439.0, "completions/mean_length": 47.625, "completions/min_length": 44.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.625, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.986204206943512, "rewards/meter/std": 0.010548613965511322, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7788560390472412, "rewards/repeat_soft/std": 0.12844043970108032, "rewards/judge_quality/mean": 0.38875001668930054, "rewards/judge_quality/std": 0.08675704896450043, "rewards/total_composite/mean": 0.788302481174469, "rewards/total_composite/std": 0.024745091795921326, "reward": 0.788302481174469, "reward_std": 0.024745095521211624, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10078133642673492, "sampling/sampling_logp_difference/max": 1.3187580108642578, "sampling/importance_sampling_ratio/min": 0.26746729016304016, "sampling/importance_sampling_ratio/mean": 1.0209405422210693, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5104928575456142, "clip_ratio/low_mean": 0.045124680269509554, "clip_ratio/low_min": 0.045124680269509554, "clip_ratio/high_mean": 0.060861500445753336, "clip_ratio/high_max": 0.060861500445753336, "clip_ratio/region_mean": 0.10598618071526289, "reward_total_mean": 0.788302481174469, "reward_meter_mean": 0.986204206943512, "reward_meter_std": 0.010548613965511322, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7788560390472412, "reward_repeat_soft_std": 0.12844043970108032, "reward_judge_quality_mean": 0.38875001668930054, "reward_judge_quality_std": 0.08675704896450043, "reward_total_composite_mean": 0.788302481174469, "reward_total_composite_std": 0.024745091795921326} {"timestamp_utc": "2026-04-13T01:10:50Z", "mode": "train", "global_step": 1155, "epoch": 0.11602209944751381, "loss": 0.0118, "grad_norm": 9.176629066467285, "learning_rate": 6.503030303030303e-06, "num_tokens": 2078143.0, "completions/mean_length": 54.0, "completions/min_length": 45.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9831566214561462, "rewards/meter/std": 0.027357593178749084, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8722337484359741, "rewards/repeat_soft/std": 0.14056901633739471, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.8255188465118408, "rewards/total_composite/std": 0.05655379965901375, "reward": 0.8255188465118408, "reward_std": 0.056553807109594345, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.129767045378685, "sampling/sampling_logp_difference/max": 1.947404146194458, "sampling/importance_sampling_ratio/min": 0.14264386892318726, "sampling/importance_sampling_ratio/mean": 1.0187934637069702, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8121622800827026, "clip_ratio/low_mean": 0.10154590895399451, "clip_ratio/low_min": 0.10154590895399451, "clip_ratio/high_mean": 0.006818181835114956, "clip_ratio/high_max": 0.006818181835114956, "clip_ratio/region_mean": 0.10836409078910947, "reward_total_mean": 0.8255188465118408, "reward_meter_mean": 0.9831566214561462, "reward_meter_std": 0.027357593178749084, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8722337484359741, "reward_repeat_soft_std": 0.14056901633739471, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.8255188465118408, "reward_total_composite_std": 0.05655379965901375} {"timestamp_utc": "2026-04-13T01:10:57Z", "mode": "train", "global_step": 1156, "epoch": 0.1161225514816675, "loss": -0.0054, "grad_norm": 10.601710319519043, "learning_rate": 6.5000000000000004e-06, "num_tokens": 2079900.0, "completions/mean_length": 50.625, "completions/min_length": 48.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.625, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.8752580285072327, "rewards/meter/std": 0.3281184136867523, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8556352853775024, "rewards/repeat_soft/std": 0.08140510320663452, "rewards/judge_quality/mean": 0.38874998688697815, "rewards/judge_quality/std": 0.08675704896450043, "rewards/total_composite/mean": 0.7460546493530273, "rewards/total_composite/std": 0.14468523859977722, "reward": 0.7460546493530273, "reward_std": 0.14468523859977722, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09351218491792679, "sampling/sampling_logp_difference/max": 1.8704984188079834, "sampling/importance_sampling_ratio/min": 0.15404686331748962, "sampling/importance_sampling_ratio/mean": 0.9932766556739807, "sampling/importance_sampling_ratio/max": 1.8117001056671143, "entropy": 0.5243814997375011, "clip_ratio/low_mean": 0.015997024020180106, "clip_ratio/low_min": 0.015997024020180106, "clip_ratio/high_mean": 0.06726290518417954, "clip_ratio/high_max": 0.06726290518417954, "clip_ratio/region_mean": 0.08325992920435965, "reward_total_mean": 0.7460546493530273, "reward_meter_mean": 0.8752580285072327, "reward_meter_std": 0.3281184136867523, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8556352853775024, "reward_repeat_soft_std": 0.08140510320663452, "reward_judge_quality_mean": 0.38874998688697815, "reward_judge_quality_std": 0.08675704896450043, "reward_total_composite_mean": 0.7460546493530273, "reward_total_composite_std": 0.14468523859977722} {"timestamp_utc": "2026-04-13T01:11:03Z", "mode": "train", "global_step": 1157, "epoch": 0.1162230035158212, "loss": 0.0445, "grad_norm": 11.883487701416016, "learning_rate": 6.496969696969697e-06, "num_tokens": 2081630.0, "completions/mean_length": 47.25, "completions/min_length": 40.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.25, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9528331756591797, "rewards/meter/std": 0.049021508544683456, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8868329524993896, "rewards/repeat_soft/std": 0.048465412110090256, "rewards/judge_quality/mean": 0.32499998807907104, "rewards/judge_quality/std": 0.13887301087379456, "rewards/total_composite/mean": 0.7649582028388977, "rewards/total_composite/std": 0.058203257620334625, "reward": 0.7649582028388977, "reward_std": 0.05820325389504433, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.131172776222229, "sampling/sampling_logp_difference/max": 1.5662832260131836, "sampling/importance_sampling_ratio/min": 0.20881988108158112, "sampling/importance_sampling_ratio/mean": 1.0127041339874268, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7767539992928505, "clip_ratio/low_mean": 0.05290520470589399, "clip_ratio/low_min": 0.05290520470589399, "clip_ratio/high_mean": 0.06875511631369591, "clip_ratio/high_max": 0.06875511631369591, "clip_ratio/region_mean": 0.1216603210195899, "reward_total_mean": 0.7649582028388977, "reward_meter_mean": 0.9528331756591797, "reward_meter_std": 0.049021508544683456, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8868329524993896, "reward_repeat_soft_std": 0.048465412110090256, "reward_judge_quality_mean": 0.32499998807907104, "reward_judge_quality_std": 0.13887301087379456, "reward_total_composite_mean": 0.7649582028388977, "reward_total_composite_std": 0.058203257620334625} {"timestamp_utc": "2026-04-13T01:11:10Z", "mode": "train", "global_step": 1158, "epoch": 0.11632345554997489, "loss": 0.0276, "grad_norm": 11.638931274414062, "learning_rate": 6.493939393939395e-06, "num_tokens": 2083540.0, "completions/mean_length": 50.75, "completions/min_length": 42.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.75, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.7064998745918274, "rewards/meter/std": 0.4005739390850067, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7934017181396484, "rewards/repeat_soft/std": 0.10214653611183167, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.7357650995254517, "rewards/total_composite/std": 0.23670685291290283, "reward": 0.7357650995254517, "reward_std": 0.23670685291290283, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11966638267040253, "sampling/sampling_logp_difference/max": 1.5049991607666016, "sampling/importance_sampling_ratio/min": 0.22201748192310333, "sampling/importance_sampling_ratio/mean": 1.0112606287002563, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7143138237297535, "clip_ratio/low_mean": 0.027075471822172403, "clip_ratio/low_min": 0.027075471822172403, "clip_ratio/high_mean": 0.08176035806536674, "clip_ratio/high_max": 0.08176035806536674, "clip_ratio/region_mean": 0.10883582988753915, "reward_total_mean": 0.7357650995254517, "reward_meter_mean": 0.7064998745918274, "reward_meter_std": 0.4005739390850067, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7934017181396484, "reward_repeat_soft_std": 0.10214653611183167, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.7357650995254517, "reward_total_composite_std": 0.23670685291290283} {"timestamp_utc": "2026-04-13T01:11:16Z", "mode": "train", "global_step": 1159, "epoch": 0.11642390758412857, "loss": 0.0205, "grad_norm": 12.211884498596191, "learning_rate": 6.490909090909091e-06, "num_tokens": 2085121.0, "completions/mean_length": 38.625, "completions/min_length": 35.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.625, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.8219901323318481, "rewards/meter/std": 0.3204042613506317, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9019528031349182, "rewards/repeat_soft/std": 0.09288962930440903, "rewards/judge_quality/mean": 0.5149999856948853, "rewards/judge_quality/std": 0.2676885426044464, "rewards/total_composite/mean": 0.6913326382637024, "rewards/total_composite/std": 0.3013235628604889, "reward": 0.6913326382637024, "reward_std": 0.3013235628604889, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12429773062467575, "sampling/sampling_logp_difference/max": 1.7296695709228516, "sampling/importance_sampling_ratio/min": 0.1773429960012436, "sampling/importance_sampling_ratio/mean": 1.0082972049713135, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.787664458155632, "clip_ratio/low_mean": 0.032135793939232826, "clip_ratio/low_min": 0.032135793939232826, "clip_ratio/high_mean": 0.08792682597413659, "clip_ratio/high_max": 0.08792682597413659, "clip_ratio/region_mean": 0.12006261991336942, "reward_total_mean": 0.6913326382637024, "reward_meter_mean": 0.8219901323318481, "reward_meter_std": 0.3204042613506317, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9019528031349182, "reward_repeat_soft_std": 0.09288962930440903, "reward_judge_quality_mean": 0.5149999856948853, "reward_judge_quality_std": 0.2676885426044464, "reward_total_composite_mean": 0.6913326382637024, "reward_total_composite_std": 0.3013235628604889} {"timestamp_utc": "2026-04-13T01:11:22Z", "mode": "train", "global_step": 1160, "epoch": 0.11652435961828227, "loss": 0.0234, "grad_norm": 17.023868560791016, "learning_rate": 6.487878787878789e-06, "num_tokens": 2086443.0, "completions/mean_length": 28.25, "completions/min_length": 17.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.25, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.6839759945869446, "rewards/meter/std": 0.33307334780693054, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9407167434692383, "rewards/repeat_soft/std": 0.046315036714076996, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.6834858655929565, "rewards/total_composite/std": 0.14821992814540863, "reward": 0.6834858655929565, "reward_std": 0.14821994304656982, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15021716058254242, "sampling/sampling_logp_difference/max": 1.6194562911987305, "sampling/importance_sampling_ratio/min": 0.19800631701946259, "sampling/importance_sampling_ratio/mean": 1.015415072441101, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8631096109747887, "clip_ratio/low_mean": 0.04056759038940072, "clip_ratio/low_min": 0.04056759038940072, "clip_ratio/high_mean": 0.04914215812459588, "clip_ratio/high_max": 0.04914215812459588, "clip_ratio/region_mean": 0.0897097485139966, "reward_total_mean": 0.6834858655929565, "reward_meter_mean": 0.6839759945869446, "reward_meter_std": 0.33307334780693054, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9407167434692383, "reward_repeat_soft_std": 0.046315036714076996, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.6834858655929565, "reward_total_composite_std": 0.14821992814540863} {"timestamp_utc": "2026-04-13T01:11:29Z", "mode": "train", "global_step": 1161, "epoch": 0.11662481165243596, "loss": 0.0207, "grad_norm": 18.399917602539062, "learning_rate": 6.484848484848485e-06, "num_tokens": 2087939.0, "completions/mean_length": 23.0, "completions/min_length": 20.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.0, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.8719699382781982, "rewards/meter/std": 0.29894304275512695, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9237529039382935, "rewards/repeat_soft/std": 0.04524531587958336, "rewards/judge_quality/mean": 0.5600000023841858, "rewards/judge_quality/std": 0.22258226573467255, "rewards/total_composite/mean": 0.8027617931365967, "rewards/total_composite/std": 0.10225704312324524, "reward": 0.8027617931365967, "reward_std": 0.10225704312324524, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14817026257514954, "sampling/sampling_logp_difference/max": 1.3629143238067627, "sampling/importance_sampling_ratio/min": 0.2559138536453247, "sampling/importance_sampling_ratio/mean": 1.0181777477264404, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.756064347922802, "clip_ratio/low_mean": 0.04880952462553978, "clip_ratio/low_min": 0.04880952462553978, "clip_ratio/high_mean": 0.13156620459631085, "clip_ratio/high_max": 0.13156620459631085, "clip_ratio/region_mean": 0.18037572922185063, "reward_total_mean": 0.8027617931365967, "reward_meter_mean": 0.8719699382781982, "reward_meter_std": 0.29894304275512695, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9237529039382935, "reward_repeat_soft_std": 0.04524531587958336, "reward_judge_quality_mean": 0.5600000023841858, "reward_judge_quality_std": 0.22258226573467255, "reward_total_composite_mean": 0.8027617931365967, "reward_total_composite_std": 0.10225704312324524} {"timestamp_utc": "2026-04-13T01:11:36Z", "mode": "train", "global_step": 1162, "epoch": 0.11672526368658966, "loss": 0.1745, "grad_norm": 11.234949111938477, "learning_rate": 6.481818181818182e-06, "num_tokens": 2089999.0, "completions/mean_length": 74.5, "completions/min_length": 53.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.5, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.8688533306121826, "rewards/meter/std": 0.2361202985048294, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9038604497909546, "rewards/repeat_soft/std": 0.04636641964316368, "rewards/judge_quality/mean": 0.41874998807907104, "rewards/judge_quality/std": 0.21931305527687073, "rewards/total_composite/mean": 0.7007450461387634, "rewards/total_composite/std": 0.12300727516412735, "reward": 0.7007450461387634, "reward_std": 0.12300727516412735, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11737444996833801, "sampling/sampling_logp_difference/max": 1.201353669166565, "sampling/importance_sampling_ratio/min": 0.3007867634296417, "sampling/importance_sampling_ratio/mean": 1.0004569292068481, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.793025054037571, "clip_ratio/low_mean": 0.04027580842375755, "clip_ratio/low_min": 0.04027580842375755, "clip_ratio/high_mean": 0.08841311652213335, "clip_ratio/high_max": 0.08841311652213335, "clip_ratio/region_mean": 0.1286889249458909, "reward_total_mean": 0.7007450461387634, "reward_meter_mean": 0.8688533306121826, "reward_meter_std": 0.2361202985048294, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9038604497909546, "reward_repeat_soft_std": 0.04636641964316368, "reward_judge_quality_mean": 0.41874998807907104, "reward_judge_quality_std": 0.21931305527687073, "reward_total_composite_mean": 0.7007450461387634, "reward_total_composite_std": 0.12300727516412735} {"timestamp_utc": "2026-04-13T01:11:42Z", "mode": "train", "global_step": 1163, "epoch": 0.11682571572074335, "loss": 0.0393, "grad_norm": 9.568799018859863, "learning_rate": 6.478787878787879e-06, "num_tokens": 2091954.0, "completions/mean_length": 68.375, "completions/min_length": 65.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.375, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.5194909572601318, "rewards/meter/std": 0.32462096214294434, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9141101837158203, "rewards/repeat_soft/std": 0.05225375294685364, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.5824319124221802, "rewards/total_composite/std": 0.16729678213596344, "reward": 0.5824319124221802, "reward_std": 0.16729679703712463, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11941255629062653, "sampling/sampling_logp_difference/max": 1.2420892715454102, "sampling/importance_sampling_ratio/min": 0.28878024220466614, "sampling/importance_sampling_ratio/mean": 1.0191059112548828, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5885546766221523, "clip_ratio/low_mean": 0.04337474098429084, "clip_ratio/low_min": 0.04337474098429084, "clip_ratio/high_mean": 0.045919747557491064, "clip_ratio/high_max": 0.045919747557491064, "clip_ratio/region_mean": 0.0892944885417819, "reward_total_mean": 0.5824319124221802, "reward_meter_mean": 0.5194909572601318, "reward_meter_std": 0.32462096214294434, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9141101837158203, "reward_repeat_soft_std": 0.05225375294685364, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.5824319124221802, "reward_total_composite_std": 0.16729678213596344} {"timestamp_utc": "2026-04-13T01:11:49Z", "mode": "train", "global_step": 1164, "epoch": 0.11692616775489703, "loss": -0.0072, "grad_norm": 16.62135124206543, "learning_rate": 6.475757575757576e-06, "num_tokens": 2093427.0, "completions/mean_length": 25.125, "completions/min_length": 24.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.125, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.9658212065696716, "rewards/meter/std": 0.04885482043027878, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.941265344619751, "rewards/repeat_soft/std": 0.06006060913205147, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8103711009025574, "rewards/total_composite/std": 0.026875976473093033, "reward": 0.8103711009025574, "reward_std": 0.026875972747802734, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1011912077665329, "sampling/sampling_logp_difference/max": 1.1830148696899414, "sampling/importance_sampling_ratio/min": 0.30635371804237366, "sampling/importance_sampling_ratio/mean": 1.0038814544677734, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6365524455904961, "clip_ratio/low_mean": 0.031249999534338713, "clip_ratio/low_min": 0.031249999534338713, "clip_ratio/high_mean": 0.07438390422612429, "clip_ratio/high_max": 0.07438390422612429, "clip_ratio/region_mean": 0.105633903760463, "reward_total_mean": 0.8103711009025574, "reward_meter_mean": 0.9658212065696716, "reward_meter_std": 0.04885482043027878, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.941265344619751, "reward_repeat_soft_std": 0.06006060913205147, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8103711009025574, "reward_total_composite_std": 0.026875976473093033} {"timestamp_utc": "2026-04-13T01:11:56Z", "mode": "train", "global_step": 1165, "epoch": 0.11702661978905073, "loss": -0.0257, "grad_norm": 8.741467475891113, "learning_rate": 6.472727272727272e-06, "num_tokens": 2095963.0, "completions/mean_length": 136.0, "completions/min_length": 111.0, "completions/max_length": 166.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.0, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 166.0, "rewards/meter/mean": 0.991001546382904, "rewards/meter/std": 0.006153270602226257, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.08908706158399582, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8470219373703003, "rewards/repeat_soft/std": 0.08810413628816605, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.756402850151062, "rewards/total_composite/std": 0.027030454948544502, "reward": 0.756402850151062, "reward_std": 0.02703046053647995, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14130398631095886, "sampling/sampling_logp_difference/max": 2.367004871368408, "sampling/importance_sampling_ratio/min": 0.0937611311674118, "sampling/importance_sampling_ratio/mean": 1.0214067697525024, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.108271636068821, "clip_ratio/low_mean": 0.058196236845105886, "clip_ratio/low_min": 0.058196236845105886, "clip_ratio/high_mean": 0.048488606698811054, "clip_ratio/high_max": 0.048488606698811054, "clip_ratio/region_mean": 0.10668484354391694, "reward_total_mean": 0.756402850151062, "reward_meter_mean": 0.991001546382904, "reward_meter_std": 0.006153270602226257, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.08908706158399582, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8470219373703003, "reward_repeat_soft_std": 0.08810413628816605, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.756402850151062, "reward_total_composite_std": 0.027030454948544502} {"timestamp_utc": "2026-04-13T01:12:07Z", "mode": "train", "global_step": 1166, "epoch": 0.11712707182320442, "loss": -0.1186, "grad_norm": 3.4389195442199707, "learning_rate": 6.4696969696969705e-06, "num_tokens": 2097673.0, "completions/mean_length": 101.75, "completions/min_length": 36.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 43.142860412597656, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.7515466213226318, "rewards/meter/std": 0.2958497703075409, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.1178511381149292, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.868537425994873, "rewards/repeat_soft/std": 0.10970394313335419, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.13715866208076477, "rewards/total_composite/mean": 0.5416536331176758, "rewards/total_composite/std": 0.25411245226860046, "reward": 0.5416536331176758, "reward_std": 0.25411245226860046, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10232638567686081, "sampling/sampling_logp_difference/max": 1.5656452178955078, "sampling/importance_sampling_ratio/min": 0.2089531421661377, "sampling/importance_sampling_ratio/mean": 1.028954267501831, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6652880311012268, "clip_ratio/low_mean": 0.049985820427536964, "clip_ratio/low_min": 0.049985820427536964, "clip_ratio/high_mean": 0.041169234085828066, "clip_ratio/high_max": 0.041169234085828066, "clip_ratio/region_mean": 0.09115505451336503, "reward_total_mean": 0.5416536331176758, "reward_meter_mean": 0.7515466213226318, "reward_meter_std": 0.2958497703075409, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.1178511381149292, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.868537425994873, "reward_repeat_soft_std": 0.10970394313335419, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.13715866208076477, "reward_total_composite_mean": 0.5416536331176758, "reward_total_composite_std": 0.25411245226860046} {"timestamp_utc": "2026-04-13T01:12:14Z", "mode": "train", "global_step": 1167, "epoch": 0.11722752385735812, "loss": 0.092, "grad_norm": 29.857378005981445, "learning_rate": 6.466666666666667e-06, "num_tokens": 2099357.0, "completions/mean_length": 37.5, "completions/min_length": 34.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.5, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.5666708946228027, "rewards/meter/std": 0.42524775862693787, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9593240022659302, "rewards/repeat_soft/std": 0.058273062109947205, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.12631452083587646, "rewards/total_composite/mean": 0.6003093123435974, "rewards/total_composite/std": 0.20172002911567688, "reward": 0.6003093123435974, "reward_std": 0.20172001421451569, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16675153374671936, "sampling/sampling_logp_difference/max": 1.3335566520690918, "sampling/importance_sampling_ratio/min": 0.26353827118873596, "sampling/importance_sampling_ratio/mean": 1.026526689529419, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2839810848236084, "clip_ratio/low_mean": 0.09596755728125572, "clip_ratio/low_min": 0.09596755728125572, "clip_ratio/high_mean": 0.07182106003165245, "clip_ratio/high_max": 0.07182106003165245, "clip_ratio/region_mean": 0.16778861731290817, "reward_total_mean": 0.6003093123435974, "reward_meter_mean": 0.5666708946228027, "reward_meter_std": 0.42524775862693787, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9593240022659302, "reward_repeat_soft_std": 0.058273062109947205, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.12631452083587646, "reward_total_composite_mean": 0.6003093123435974, "reward_total_composite_std": 0.20172002911567688} {"timestamp_utc": "2026-04-13T01:12:20Z", "mode": "train", "global_step": 1168, "epoch": 0.11732797589151181, "loss": -0.0146, "grad_norm": 17.309467315673828, "learning_rate": 6.463636363636364e-06, "num_tokens": 2101098.0, "completions/mean_length": 47.625, "completions/min_length": 41.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.625, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.8277602195739746, "rewards/meter/std": 0.33753395080566406, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9358220100402832, "rewards/repeat_soft/std": 0.04466966539621353, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404937028885, "rewards/total_composite/mean": 0.7829493284225464, "rewards/total_composite/std": 0.10631219297647476, "reward": 0.7829493284225464, "reward_std": 0.10631217807531357, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14641635119915009, "sampling/sampling_logp_difference/max": 1.7511992454528809, "sampling/importance_sampling_ratio/min": 0.17356567084789276, "sampling/importance_sampling_ratio/mean": 1.011094331741333, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8895816057920456, "clip_ratio/low_mean": 0.03187963832169771, "clip_ratio/low_min": 0.03187963832169771, "clip_ratio/high_mean": 0.09048813860863447, "clip_ratio/high_max": 0.09048813860863447, "clip_ratio/region_mean": 0.12236777693033218, "reward_total_mean": 0.7829493284225464, "reward_meter_mean": 0.8277602195739746, "reward_meter_std": 0.33753395080566406, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9358220100402832, "reward_repeat_soft_std": 0.04466966539621353, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404937028885, "reward_total_composite_mean": 0.7829493284225464, "reward_total_composite_std": 0.10631219297647476} {"timestamp_utc": "2026-04-13T01:12:26Z", "mode": "train", "global_step": 1169, "epoch": 0.11742842792566549, "loss": 0.0244, "grad_norm": 10.435402870178223, "learning_rate": 6.460606060606061e-06, "num_tokens": 2102831.0, "completions/mean_length": 53.625, "completions/min_length": 51.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.625, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9841208457946777, "rewards/meter/std": 0.017493046820163727, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9488693475723267, "rewards/repeat_soft/std": 0.04408172145485878, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.1011011004447937, "rewards/total_composite/mean": 0.8294912576675415, "rewards/total_composite/std": 0.02979624830186367, "reward": 0.8294912576675415, "reward_std": 0.02979625016450882, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1532936990261078, "sampling/sampling_logp_difference/max": 2.035396099090576, "sampling/importance_sampling_ratio/min": 0.13062873482704163, "sampling/importance_sampling_ratio/mean": 1.0092501640319824, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9839647859334946, "clip_ratio/low_mean": 0.06852115504443645, "clip_ratio/low_min": 0.06852115504443645, "clip_ratio/high_mean": 0.0526433065533638, "clip_ratio/high_max": 0.0526433065533638, "clip_ratio/region_mean": 0.12116446159780025, "reward_total_mean": 0.8294912576675415, "reward_meter_mean": 0.9841208457946777, "reward_meter_std": 0.017493046820163727, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9488693475723267, "reward_repeat_soft_std": 0.04408172145485878, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.1011011004447937, "reward_total_composite_mean": 0.8294912576675415, "reward_total_composite_std": 0.02979624830186367} {"timestamp_utc": "2026-04-13T01:12:33Z", "mode": "train", "global_step": 1170, "epoch": 0.11752887995981919, "loss": -0.0449, "grad_norm": 11.702943801879883, "learning_rate": 6.457575757575758e-06, "num_tokens": 2104455.0, "completions/mean_length": 47.0, "completions/min_length": 38.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.0, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.498923659324646, "rewards/meter/std": 0.35543274879455566, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9547202587127686, "rewards/repeat_soft/std": 0.030370263382792473, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.25150617957115173, "rewards/total_composite/mean": 0.5510541200637817, "rewards/total_composite/std": 0.2703082859516144, "reward": 0.5510541200637817, "reward_std": 0.27030831575393677, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13220873475074768, "sampling/sampling_logp_difference/max": 1.7129364013671875, "sampling/importance_sampling_ratio/min": 0.18033547699451447, "sampling/importance_sampling_ratio/mean": 1.0074419975280762, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9633477702736855, "clip_ratio/low_mean": 0.03609219752252102, "clip_ratio/low_min": 0.03609219752252102, "clip_ratio/high_mean": 0.08219537604600191, "clip_ratio/high_max": 0.08219537604600191, "clip_ratio/region_mean": 0.11828757356852293, "reward_total_mean": 0.5510541200637817, "reward_meter_mean": 0.498923659324646, "reward_meter_std": 0.35543274879455566, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9547202587127686, "reward_repeat_soft_std": 0.030370263382792473, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.25150617957115173, "reward_total_composite_mean": 0.5510541200637817, "reward_total_composite_std": 0.2703082859516144} {"timestamp_utc": "2026-04-13T01:12:44Z", "mode": "train", "global_step": 1171, "epoch": 0.11762933199397288, "loss": -0.1102, "grad_norm": 3.2834084033966064, "learning_rate": 6.454545454545456e-06, "num_tokens": 2106078.0, "completions/mean_length": 99.875, "completions/min_length": 37.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 41.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.855064868927002, "rewards/meter/std": 0.24616745114326477, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9318563938140869, "rewards/repeat_soft/std": 0.05220234394073486, "rewards/judge_quality/mean": 0.5112500190734863, "rewards/judge_quality/std": 0.268936425447464, "rewards/total_composite/mean": 0.7719647884368896, "rewards/total_composite/std": 0.19638890027999878, "reward": 0.7719647884368896, "reward_std": 0.1963888704776764, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1262025535106659, "sampling/sampling_logp_difference/max": 1.579547643661499, "sampling/importance_sampling_ratio/min": 0.20606829226016998, "sampling/importance_sampling_ratio/mean": 0.9935263395309448, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7497023344039917, "clip_ratio/low_mean": 0.0031250000465661287, "clip_ratio/low_min": 0.0031250000465661287, "clip_ratio/high_mean": 0.11108048353344202, "clip_ratio/high_max": 0.11108048353344202, "clip_ratio/region_mean": 0.11420548358000815, "reward_total_mean": 0.7719647884368896, "reward_meter_mean": 0.855064868927002, "reward_meter_std": 0.24616745114326477, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9318563938140869, "reward_repeat_soft_std": 0.05220234394073486, "reward_judge_quality_mean": 0.5112500190734863, "reward_judge_quality_std": 0.268936425447464, "reward_total_composite_mean": 0.7719647884368896, "reward_total_composite_std": 0.19638890027999878} {"timestamp_utc": "2026-04-13T01:12:51Z", "mode": "train", "global_step": 1172, "epoch": 0.11772978402812657, "loss": -0.1013, "grad_norm": 9.810096740722656, "learning_rate": 6.451515151515152e-06, "num_tokens": 2108098.0, "completions/mean_length": 72.5, "completions/min_length": 45.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.5, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.98907470703125, "rewards/meter/std": 0.007043017074465752, "rewards/count_adherence/mean": 0.71875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7902984619140625, "rewards/repeat_soft/std": 0.20483694970607758, "rewards/judge_quality/mean": 0.39375001192092896, "rewards/judge_quality/std": 0.15638209879398346, "rewards/total_composite/mean": 0.750050961971283, "rewards/total_composite/std": 0.07179023325443268, "reward": 0.750050961971283, "reward_std": 0.07179022580385208, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12329795956611633, "sampling/sampling_logp_difference/max": 2.258842945098877, "sampling/importance_sampling_ratio/min": 0.10447128862142563, "sampling/importance_sampling_ratio/mean": 1.0046751499176025, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6439968310296535, "clip_ratio/low_mean": 0.02461484679952264, "clip_ratio/low_min": 0.02461484679952264, "clip_ratio/high_mean": 0.07978084404021502, "clip_ratio/high_max": 0.07978084404021502, "clip_ratio/region_mean": 0.10439569083973765, "reward_total_mean": 0.750050961971283, "reward_meter_mean": 0.98907470703125, "reward_meter_std": 0.007043017074465752, "reward_count_adherence_mean": 0.71875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7902984619140625, "reward_repeat_soft_std": 0.20483694970607758, "reward_judge_quality_mean": 0.39375001192092896, "reward_judge_quality_std": 0.15638209879398346, "reward_total_composite_mean": 0.750050961971283, "reward_total_composite_std": 0.07179023325443268} {"timestamp_utc": "2026-04-13T01:12:57Z", "mode": "train", "global_step": 1173, "epoch": 0.11783023606228026, "loss": 0.059, "grad_norm": 16.242639541625977, "learning_rate": 6.4484848484848496e-06, "num_tokens": 2109667.0, "completions/mean_length": 32.125, "completions/min_length": 27.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.125, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.5791162252426147, "rewards/meter/std": 0.34243446588516235, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9298205375671387, "rewards/repeat_soft/std": 0.03505006432533264, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.7420843839645386, "rewards/total_composite/std": 0.17756107449531555, "reward": 0.7420843839645386, "reward_std": 0.17756104469299316, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10038980841636658, "sampling/sampling_logp_difference/max": 1.3182693719863892, "sampling/importance_sampling_ratio/min": 0.2675980031490326, "sampling/importance_sampling_ratio/mean": 1.0138444900512695, "sampling/importance_sampling_ratio/max": 1.8715139627456665, "entropy": 0.45306534320116043, "clip_ratio/low_mean": 0.03656735457479954, "clip_ratio/low_min": 0.03656735457479954, "clip_ratio/high_mean": 0.0780990719795227, "clip_ratio/high_max": 0.0780990719795227, "clip_ratio/region_mean": 0.11466642655432224, "reward_total_mean": 0.7420843839645386, "reward_meter_mean": 0.5791162252426147, "reward_meter_std": 0.34243446588516235, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9298205375671387, "reward_repeat_soft_std": 0.03505006432533264, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.7420843839645386, "reward_total_composite_std": 0.17756107449531555} {"timestamp_utc": "2026-04-13T01:13:09Z", "mode": "train", "global_step": 1174, "epoch": 0.11793068809643395, "loss": -0.1165, "grad_norm": 3.11110782623291, "learning_rate": 6.445454545454546e-06, "num_tokens": 2111326.0, "completions/mean_length": 107.375, "completions/min_length": 43.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 49.57143020629883, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.7533804178237915, "rewards/meter/std": 0.37714123725891113, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.951667070388794, "rewards/repeat_soft/std": 0.04661647602915764, "rewards/judge_quality/mean": 0.5112500190734863, "rewards/judge_quality/std": 0.268936425447464, "rewards/total_composite/mean": 0.7006878852844238, "rewards/total_composite/std": 0.31562507152557373, "reward": 0.7006878852844238, "reward_std": 0.31562507152557373, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13108034431934357, "sampling/sampling_logp_difference/max": 1.4687743186950684, "sampling/importance_sampling_ratio/min": 0.23020747303962708, "sampling/importance_sampling_ratio/mean": 0.9850382804870605, "sampling/importance_sampling_ratio/max": 1.7451790571212769, "entropy": 0.7706992775201797, "clip_ratio/low_mean": 0.018894830718636513, "clip_ratio/low_min": 0.018894830718636513, "clip_ratio/high_mean": 0.09294932708144188, "clip_ratio/high_max": 0.09294932708144188, "clip_ratio/region_mean": 0.11184415780007839, "reward_total_mean": 0.7006878852844238, "reward_meter_mean": 0.7533804178237915, "reward_meter_std": 0.37714123725891113, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.951667070388794, "reward_repeat_soft_std": 0.04661647602915764, "reward_judge_quality_mean": 0.5112500190734863, "reward_judge_quality_std": 0.268936425447464, "reward_total_composite_mean": 0.7006878852844238, "reward_total_composite_std": 0.31562507152557373} {"timestamp_utc": "2026-04-13T01:13:17Z", "mode": "train", "global_step": 1175, "epoch": 0.11803114013058764, "loss": 0.0692, "grad_norm": 17.290369033813477, "learning_rate": 6.442424242424243e-06, "num_tokens": 2114018.0, "completions/mean_length": 118.5, "completions/min_length": 107.0, "completions/max_length": 139.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.5, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.660799503326416, "rewards/meter/std": 0.35144859552383423, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9145506620407104, "rewards/repeat_soft/std": 0.04454375430941582, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.12351980805397034, "rewards/total_composite/mean": 0.6068148612976074, "rewards/total_composite/std": 0.17081409692764282, "reward": 0.6068148612976074, "reward_std": 0.17081409692764282, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13706128299236298, "sampling/sampling_logp_difference/max": 2.084315299987793, "sampling/importance_sampling_ratio/min": 0.12439225614070892, "sampling/importance_sampling_ratio/mean": 1.0058815479278564, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.952688068151474, "clip_ratio/low_mean": 0.03445187769830227, "clip_ratio/low_min": 0.03445187769830227, "clip_ratio/high_mean": 0.10490310192108154, "clip_ratio/high_max": 0.10490310192108154, "clip_ratio/region_mean": 0.1393549796193838, "reward_total_mean": 0.6068148612976074, "reward_meter_mean": 0.660799503326416, "reward_meter_std": 0.35144859552383423, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9145506620407104, "reward_repeat_soft_std": 0.04454375430941582, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.12351980805397034, "reward_total_composite_mean": 0.6068148612976074, "reward_total_composite_std": 0.17081409692764282} {"timestamp_utc": "2026-04-13T01:13:23Z", "mode": "train", "global_step": 1176, "epoch": 0.11813159216474134, "loss": -0.0732, "grad_norm": 14.977752685546875, "learning_rate": 6.43939393939394e-06, "num_tokens": 2115483.0, "completions/mean_length": 24.125, "completions/min_length": 20.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.125, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.8203218579292297, "rewards/meter/std": 0.3332087993621826, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9418370127677917, "rewards/repeat_soft/std": 0.03792871534824371, "rewards/judge_quality/mean": 0.5637500286102295, "rewards/judge_quality/std": 0.22012579441070557, "rewards/total_composite/mean": 0.7824535369873047, "rewards/total_composite/std": 0.17672757804393768, "reward": 0.7824535369873047, "reward_std": 0.17672757804393768, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13816271722316742, "sampling/sampling_logp_difference/max": 1.4588332176208496, "sampling/importance_sampling_ratio/min": 0.23250740766525269, "sampling/importance_sampling_ratio/mean": 1.0113455057144165, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9119000434875488, "clip_ratio/low_mean": 0.08138112165033817, "clip_ratio/low_min": 0.08138112165033817, "clip_ratio/high_mean": 0.09904244262725115, "clip_ratio/high_max": 0.09904244262725115, "clip_ratio/region_mean": 0.18042356427758932, "reward_total_mean": 0.7824535369873047, "reward_meter_mean": 0.8203218579292297, "reward_meter_std": 0.3332087993621826, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9418370127677917, "reward_repeat_soft_std": 0.03792871534824371, "reward_judge_quality_mean": 0.5637500286102295, "reward_judge_quality_std": 0.22012579441070557, "reward_total_composite_mean": 0.7824535369873047, "reward_total_composite_std": 0.17672757804393768} {"timestamp_utc": "2026-04-13T01:13:29Z", "mode": "train", "global_step": 1177, "epoch": 0.11823204419889503, "loss": 0.0397, "grad_norm": 17.991966247558594, "learning_rate": 6.436363636363637e-06, "num_tokens": 2116892.0, "completions/mean_length": 22.125, "completions/min_length": 16.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.125, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.17356745898723602, "rewards/meter/std": 0.26819491386413574, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9226500391960144, "rewards/repeat_soft/std": 0.04833726957440376, "rewards/judge_quality/mean": 0.41749998927116394, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.4456203579902649, "rewards/total_composite/std": 0.12828563153743744, "reward": 0.4456203579902649, "reward_std": 0.12828563153743744, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1802975982427597, "sampling/sampling_logp_difference/max": 2.0709753036499023, "sampling/importance_sampling_ratio/min": 0.1260627806186676, "sampling/importance_sampling_ratio/mean": 1.0203771591186523, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2447180524468422, "clip_ratio/low_mean": 0.14846380800008774, "clip_ratio/low_min": 0.14846380800008774, "clip_ratio/high_mean": 0.059761904180049896, "clip_ratio/high_max": 0.059761904180049896, "clip_ratio/region_mean": 0.20822571218013763, "reward_total_mean": 0.4456203579902649, "reward_meter_mean": 0.17356745898723602, "reward_meter_std": 0.26819491386413574, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9226500391960144, "reward_repeat_soft_std": 0.04833726957440376, "reward_judge_quality_mean": 0.41749998927116394, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.4456203579902649, "reward_total_composite_std": 0.12828563153743744} {"timestamp_utc": "2026-04-13T01:13:35Z", "mode": "train", "global_step": 1178, "epoch": 0.11833249623304871, "loss": 0.0471, "grad_norm": 13.178162574768066, "learning_rate": 6.433333333333333e-06, "num_tokens": 2118641.0, "completions/mean_length": 49.625, "completions/min_length": 43.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.625, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.5941541194915771, "rewards/meter/std": 0.3054586946964264, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9581353664398193, "rewards/repeat_soft/std": 0.01717033050954342, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.6129329204559326, "rewards/total_composite/std": 0.13342300057411194, "reward": 0.6129329204559326, "reward_std": 0.13342300057411194, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.135335773229599, "sampling/sampling_logp_difference/max": 2.2560479640960693, "sampling/importance_sampling_ratio/min": 0.1047637015581131, "sampling/importance_sampling_ratio/mean": 0.9858918190002441, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6558291539549828, "clip_ratio/low_mean": 0.055285634472966194, "clip_ratio/low_min": 0.055285634472966194, "clip_ratio/high_mean": 0.06719428207725286, "clip_ratio/high_max": 0.06719428207725286, "clip_ratio/region_mean": 0.12247991655021906, "reward_total_mean": 0.6129329204559326, "reward_meter_mean": 0.5941541194915771, "reward_meter_std": 0.3054586946964264, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9581353664398193, "reward_repeat_soft_std": 0.01717033050954342, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.6129329204559326, "reward_total_composite_std": 0.13342300057411194} {"timestamp_utc": "2026-04-13T01:13:43Z", "mode": "train", "global_step": 1179, "epoch": 0.11843294826720241, "loss": 0.0456, "grad_norm": 14.437512397766113, "learning_rate": 6.430303030303031e-06, "num_tokens": 2120339.0, "completions/mean_length": 44.25, "completions/min_length": 41.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.25, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.9910371899604797, "rewards/meter/std": 0.007195891346782446, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8340160250663757, "rewards/repeat_soft/std": 0.0904294103384018, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8027433156967163, "rewards/total_composite/std": 0.024928018450737, "reward": 0.8027433156967163, "reward_std": 0.024927999824285507, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10591669380664825, "sampling/sampling_logp_difference/max": 1.0784322023391724, "sampling/importance_sampling_ratio/min": 0.3401283621788025, "sampling/importance_sampling_ratio/mean": 1.0113180875778198, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6561705805361271, "clip_ratio/low_mean": 0.03210552502423525, "clip_ratio/low_min": 0.03210552502423525, "clip_ratio/high_mean": 0.08788872696459293, "clip_ratio/high_max": 0.08788872696459293, "clip_ratio/region_mean": 0.11999425198882818, "reward_total_mean": 0.8027433156967163, "reward_meter_mean": 0.9910371899604797, "reward_meter_std": 0.007195891346782446, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8340160250663757, "reward_repeat_soft_std": 0.0904294103384018, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8027433156967163, "reward_total_composite_std": 0.024928018450737} {"timestamp_utc": "2026-04-13T01:13:49Z", "mode": "train", "global_step": 1180, "epoch": 0.1185334003013561, "loss": 0.047, "grad_norm": 13.533049583435059, "learning_rate": 6.427272727272728e-06, "num_tokens": 2122006.0, "completions/mean_length": 45.375, "completions/min_length": 43.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.375, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.9080929756164551, "rewards/meter/std": 0.1449466496706009, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9553428888320923, "rewards/repeat_soft/std": 0.03691984340548515, "rewards/judge_quality/mean": 0.3725000023841858, "rewards/judge_quality/std": 0.11055056750774384, "rewards/total_composite/mean": 0.7659261226654053, "rewards/total_composite/std": 0.09921564161777496, "reward": 0.7659261226654053, "reward_std": 0.09921563416719437, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1446894258260727, "sampling/sampling_logp_difference/max": 1.4543476104736328, "sampling/importance_sampling_ratio/min": 0.2335526943206787, "sampling/importance_sampling_ratio/mean": 1.0156254768371582, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.175966463983059, "clip_ratio/low_mean": 0.022187945432960987, "clip_ratio/low_min": 0.022187945432960987, "clip_ratio/high_mean": 0.0835075993090868, "clip_ratio/high_max": 0.0835075993090868, "clip_ratio/region_mean": 0.10569554474204779, "reward_total_mean": 0.7659261226654053, "reward_meter_mean": 0.9080929756164551, "reward_meter_std": 0.1449466496706009, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9553428888320923, "reward_repeat_soft_std": 0.03691984340548515, "reward_judge_quality_mean": 0.3725000023841858, "reward_judge_quality_std": 0.11055056750774384, "reward_total_composite_mean": 0.7659261226654053, "reward_total_composite_std": 0.09921564161777496} {"timestamp_utc": "2026-04-13T01:13:55Z", "mode": "train", "global_step": 1181, "epoch": 0.1186338523355098, "loss": 0.0208, "grad_norm": 9.545382499694824, "learning_rate": 6.424242424242425e-06, "num_tokens": 2124430.0, "completions/mean_length": 81.0, "completions/min_length": 77.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.0, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.9953722953796387, "rewards/meter/std": 0.0009534741984680295, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.843069851398468, "rewards/repeat_soft/std": 0.0678296610713005, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.7452245354652405, "rewards/total_composite/std": 0.028635336086153984, "reward": 0.7452245354652405, "reward_std": 0.028635336086153984, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10489709675312042, "sampling/sampling_logp_difference/max": 1.5083308219909668, "sampling/importance_sampling_ratio/min": 0.22127902507781982, "sampling/importance_sampling_ratio/mean": 1.0195977687835693, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5453138910233974, "clip_ratio/low_mean": 0.04677965119481087, "clip_ratio/low_min": 0.04677965119481087, "clip_ratio/high_mean": 0.04134166054427624, "clip_ratio/high_max": 0.04134166054427624, "clip_ratio/region_mean": 0.0881213117390871, "reward_total_mean": 0.7452245354652405, "reward_meter_mean": 0.9953722953796387, "reward_meter_std": 0.0009534741984680295, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.843069851398468, "reward_repeat_soft_std": 0.0678296610713005, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.7452245354652405, "reward_total_composite_std": 0.028635336086153984} {"timestamp_utc": "2026-04-13T01:14:01Z", "mode": "train", "global_step": 1182, "epoch": 0.11873430436966348, "loss": 0.0465, "grad_norm": 23.564491271972656, "learning_rate": 6.4212121212121215e-06, "num_tokens": 2125769.0, "completions/mean_length": 22.375, "completions/min_length": 19.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.375, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.7413561344146729, "rewards/meter/std": 0.45080897212028503, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8936620950698853, "rewards/repeat_soft/std": 0.07387591153383255, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.0843462198972702, "rewards/total_composite/mean": 0.6884764432907104, "rewards/total_composite/std": 0.21004797518253326, "reward": 0.6884764432907104, "reward_std": 0.21004796028137207, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1533241719007492, "sampling/sampling_logp_difference/max": 2.518807888031006, "sampling/importance_sampling_ratio/min": 0.08055558055639267, "sampling/importance_sampling_ratio/mean": 0.9981934428215027, "sampling/importance_sampling_ratio/max": 1.7040468454360962, "entropy": 0.9394619464874268, "clip_ratio/low_mean": 0.047727273777127266, "clip_ratio/low_min": 0.047727273777127266, "clip_ratio/high_mean": 0.11219415534287691, "clip_ratio/high_max": 0.11219415534287691, "clip_ratio/region_mean": 0.15992142912000418, "reward_total_mean": 0.6884764432907104, "reward_meter_mean": 0.7413561344146729, "reward_meter_std": 0.45080897212028503, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8936620950698853, "reward_repeat_soft_std": 0.07387591153383255, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.0843462198972702, "reward_total_composite_mean": 0.6884764432907104, "reward_total_composite_std": 0.21004797518253326} {"timestamp_utc": "2026-04-13T01:14:07Z", "mode": "train", "global_step": 1183, "epoch": 0.11883475640381717, "loss": 0.099, "grad_norm": 15.640925407409668, "learning_rate": 6.418181818181819e-06, "num_tokens": 2127423.0, "completions/mean_length": 30.75, "completions/min_length": 23.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.75, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.44571012258529663, "rewards/meter/std": 0.38221052289009094, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9408128261566162, "rewards/repeat_soft/std": 0.02289712429046631, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.19334925711154938, "rewards/total_composite/mean": 0.5852758288383484, "rewards/total_composite/std": 0.19092252850532532, "reward": 0.5852758288383484, "reward_std": 0.19092251360416412, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13780151307582855, "sampling/sampling_logp_difference/max": 1.886746883392334, "sampling/importance_sampling_ratio/min": 0.1515640765428543, "sampling/importance_sampling_ratio/mean": 1.0249344110488892, "sampling/importance_sampling_ratio/max": 1.9846606254577637, "entropy": 0.7380145490169525, "clip_ratio/low_mean": 0.07175275217741728, "clip_ratio/low_min": 0.07175275217741728, "clip_ratio/high_mean": 0.05841904226690531, "clip_ratio/high_max": 0.05841904226690531, "clip_ratio/region_mean": 0.13017179444432259, "reward_total_mean": 0.5852758288383484, "reward_meter_mean": 0.44571012258529663, "reward_meter_std": 0.38221052289009094, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9408128261566162, "reward_repeat_soft_std": 0.02289712429046631, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.19334925711154938, "reward_total_composite_mean": 0.5852758288383484, "reward_total_composite_std": 0.19092252850532532} {"timestamp_utc": "2026-04-13T01:14:14Z", "mode": "train", "global_step": 1184, "epoch": 0.11893520843797087, "loss": 0.0485, "grad_norm": 9.340415000915527, "learning_rate": 6.415151515151515e-06, "num_tokens": 2129115.0, "completions/mean_length": 46.5, "completions/min_length": 44.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.5, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.6359912157058716, "rewards/meter/std": 0.408885657787323, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9168011546134949, "rewards/repeat_soft/std": 0.05564064159989357, "rewards/judge_quality/mean": 0.5862500071525574, "rewards/judge_quality/std": 0.2353682667016983, "rewards/total_composite/mean": 0.7037512063980103, "rewards/total_composite/std": 0.18152360618114471, "reward": 0.7037512063980103, "reward_std": 0.18152360618114471, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12934625148773193, "sampling/sampling_logp_difference/max": 2.0269484519958496, "sampling/importance_sampling_ratio/min": 0.13173691928386688, "sampling/importance_sampling_ratio/mean": 0.9931597113609314, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5227286219596863, "clip_ratio/low_mean": 0.04581363219767809, "clip_ratio/low_min": 0.04581363219767809, "clip_ratio/high_mean": 0.0719580166041851, "clip_ratio/high_max": 0.0719580166041851, "clip_ratio/region_mean": 0.1177716488018632, "reward_total_mean": 0.7037512063980103, "reward_meter_mean": 0.6359912157058716, "reward_meter_std": 0.408885657787323, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9168011546134949, "reward_repeat_soft_std": 0.05564064159989357, "reward_judge_quality_mean": 0.5862500071525574, "reward_judge_quality_std": 0.2353682667016983, "reward_total_composite_mean": 0.7037512063980103, "reward_total_composite_std": 0.18152360618114471} {"timestamp_utc": "2026-04-13T01:14:20Z", "mode": "train", "global_step": 1185, "epoch": 0.11903566047212456, "loss": 0.0455, "grad_norm": 10.517217636108398, "learning_rate": 6.412121212121213e-06, "num_tokens": 2130695.0, "completions/mean_length": 45.5, "completions/min_length": 36.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.984275221824646, "rewards/meter/std": 0.01585344411432743, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.943330705165863, "rewards/repeat_soft/std": 0.04798226058483124, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.22403763234615326, "rewards/total_composite/mean": 0.8155069351196289, "rewards/total_composite/std": 0.0674963891506195, "reward": 0.8155069351196289, "reward_std": 0.06749638170003891, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13911187648773193, "sampling/sampling_logp_difference/max": 2.3968522548675537, "sampling/importance_sampling_ratio/min": 0.09100396186113358, "sampling/importance_sampling_ratio/mean": 1.0058107376098633, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.905344270169735, "clip_ratio/low_mean": 0.046417233534157276, "clip_ratio/low_min": 0.046417233534157276, "clip_ratio/high_mean": 0.09576424676924944, "clip_ratio/high_max": 0.09576424676924944, "clip_ratio/region_mean": 0.14218148030340672, "reward_total_mean": 0.8155069351196289, "reward_meter_mean": 0.984275221824646, "reward_meter_std": 0.01585344411432743, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.943330705165863, "reward_repeat_soft_std": 0.04798226058483124, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.22403763234615326, "reward_total_composite_mean": 0.8155069351196289, "reward_total_composite_std": 0.0674963891506195} {"timestamp_utc": "2026-04-13T01:14:27Z", "mode": "train", "global_step": 1186, "epoch": 0.11913611250627826, "loss": 0.0836, "grad_norm": 11.133027076721191, "learning_rate": 6.40909090909091e-06, "num_tokens": 2132419.0, "completions/mean_length": 53.5, "completions/min_length": 49.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.5, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9862897396087646, "rewards/meter/std": 0.017070094123482704, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9643714427947998, "rewards/repeat_soft/std": 0.04347429424524307, "rewards/judge_quality/mean": 0.5099999904632568, "rewards/judge_quality/std": 0.1302744597196579, "rewards/total_composite/mean": 0.843267560005188, "rewards/total_composite/std": 0.04268180951476097, "reward": 0.843267560005188, "reward_std": 0.04268180951476097, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12370169162750244, "sampling/sampling_logp_difference/max": 1.3131130933761597, "sampling/importance_sampling_ratio/min": 0.268981397151947, "sampling/importance_sampling_ratio/mean": 1.0193146467208862, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0122430473566055, "clip_ratio/low_mean": 0.08684440981596708, "clip_ratio/low_min": 0.08684440981596708, "clip_ratio/high_mean": 0.025255102664232254, "clip_ratio/high_max": 0.025255102664232254, "clip_ratio/region_mean": 0.11209951248019934, "reward_total_mean": 0.843267560005188, "reward_meter_mean": 0.9862897396087646, "reward_meter_std": 0.017070094123482704, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9643714427947998, "reward_repeat_soft_std": 0.04347429424524307, "reward_judge_quality_mean": 0.5099999904632568, "reward_judge_quality_std": 0.1302744597196579, "reward_total_composite_mean": 0.843267560005188, "reward_total_composite_std": 0.04268180951476097} {"timestamp_utc": "2026-04-13T01:14:40Z", "mode": "train", "global_step": 1187, "epoch": 0.11923656454043194, "loss": -0.0519, "grad_norm": 16.63850212097168, "learning_rate": 6.406060606060607e-06, "num_tokens": 2134052.0, "completions/mean_length": 101.125, "completions/min_length": 35.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 42.42857360839844, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.18229788541793823, "rewards/meter/std": 0.33859315514564514, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.1178511381149292, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9732555747032166, "rewards/repeat_soft/std": 0.027958445250988007, "rewards/judge_quality/mean": 0.3387500047683716, "rewards/judge_quality/std": 0.14327171444892883, "rewards/total_composite/mean": 0.3540608882904053, "rewards/total_composite/std": 0.21620765328407288, "reward": 0.3540608882904053, "reward_std": 0.21620763838291168, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17325741052627563, "sampling/sampling_logp_difference/max": 1.6850348711013794, "sampling/importance_sampling_ratio/min": 0.18543796241283417, "sampling/importance_sampling_ratio/mean": 1.005304217338562, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.98820461332798, "clip_ratio/low_mean": 0.07492499053478241, "clip_ratio/low_min": 0.07492499053478241, "clip_ratio/high_mean": 0.06881549023091793, "clip_ratio/high_max": 0.06881549023091793, "clip_ratio/region_mean": 0.14374048076570034, "reward_total_mean": 0.3540608882904053, "reward_meter_mean": 0.18229788541793823, "reward_meter_std": 0.33859315514564514, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.1178511381149292, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9732555747032166, "reward_repeat_soft_std": 0.027958445250988007, "reward_judge_quality_mean": 0.3387500047683716, "reward_judge_quality_std": 0.14327171444892883, "reward_total_composite_mean": 0.3540608882904053, "reward_total_composite_std": 0.21620765328407288} {"timestamp_utc": "2026-04-13T01:14:46Z", "mode": "train", "global_step": 1188, "epoch": 0.11933701657458563, "loss": 0.0106, "grad_norm": 9.197784423828125, "learning_rate": 6.403030303030303e-06, "num_tokens": 2135938.0, "completions/mean_length": 54.75, "completions/min_length": 50.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.75, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.983067512512207, "rewards/meter/std": 0.02504155971109867, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9262603521347046, "rewards/repeat_soft/std": 0.05437895283102989, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.8308814167976379, "rewards/total_composite/std": 0.0551200695335865, "reward": 0.8308814167976379, "reward_std": 0.05512005090713501, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11670330166816711, "sampling/sampling_logp_difference/max": 1.7402019500732422, "sampling/importance_sampling_ratio/min": 0.17548495531082153, "sampling/importance_sampling_ratio/mean": 1.0106641054153442, "sampling/importance_sampling_ratio/max": 1.8312041759490967, "entropy": 0.7233670726418495, "clip_ratio/low_mean": 0.10226533561944962, "clip_ratio/low_min": 0.10226533561944962, "clip_ratio/high_mean": 0.006818181835114956, "clip_ratio/high_max": 0.006818181835114956, "clip_ratio/region_mean": 0.10908351745456457, "reward_total_mean": 0.8308814167976379, "reward_meter_mean": 0.983067512512207, "reward_meter_std": 0.02504155971109867, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9262603521347046, "reward_repeat_soft_std": 0.05437895283102989, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.8308814167976379, "reward_total_composite_std": 0.0551200695335865} {"timestamp_utc": "2026-04-13T01:14:54Z", "mode": "train", "global_step": 1189, "epoch": 0.11943746860873933, "loss": 0.0495, "grad_norm": 10.223784446716309, "learning_rate": 6.4000000000000006e-06, "num_tokens": 2137608.0, "completions/mean_length": 51.75, "completions/min_length": 47.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.75, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9927749633789062, "rewards/meter/std": 0.0016291373176500201, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9719986319541931, "rewards/repeat_soft/std": 0.02981742098927498, "rewards/judge_quality/mean": 0.5362499952316284, "rewards/judge_quality/std": 0.1524970978498459, "rewards/total_composite/mean": 0.8548235893249512, "rewards/total_composite/std": 0.04616691917181015, "reward": 0.8548235893249512, "reward_std": 0.04616692662239075, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13308431208133698, "sampling/sampling_logp_difference/max": 1.617020606994629, "sampling/importance_sampling_ratio/min": 0.1984892040491104, "sampling/importance_sampling_ratio/mean": 1.0164716243743896, "sampling/importance_sampling_ratio/max": 1.826738953590393, "entropy": 0.8237783163785934, "clip_ratio/low_mean": 0.09639855660498142, "clip_ratio/low_min": 0.09639855660498142, "clip_ratio/high_mean": 0.05313625279814005, "clip_ratio/high_max": 0.05313625279814005, "clip_ratio/region_mean": 0.14953480940312147, "reward_total_mean": 0.8548235893249512, "reward_meter_mean": 0.9927749633789062, "reward_meter_std": 0.0016291373176500201, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9719986319541931, "reward_repeat_soft_std": 0.02981742098927498, "reward_judge_quality_mean": 0.5362499952316284, "reward_judge_quality_std": 0.1524970978498459, "reward_total_composite_mean": 0.8548235893249512, "reward_total_composite_std": 0.04616691917181015} {"timestamp_utc": "2026-04-13T01:15:00Z", "mode": "train", "global_step": 1190, "epoch": 0.11953792064289302, "loss": 0.0335, "grad_norm": 12.107315063476562, "learning_rate": 6.396969696969697e-06, "num_tokens": 2139261.0, "completions/mean_length": 52.625, "completions/min_length": 49.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.625, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.987539529800415, "rewards/meter/std": 0.020233290269970894, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9386054277420044, "rewards/repeat_soft/std": 0.08637025207281113, "rewards/judge_quality/mean": 0.8949999809265137, "rewards/judge_quality/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9567533135414124, "rewards/total_composite/std": 0.02130885049700737, "reward": 0.9567533135414124, "reward_std": 0.021308863535523415, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12019576877355576, "sampling/sampling_logp_difference/max": 2.44616436958313, "sampling/importance_sampling_ratio/min": 0.08662521094083786, "sampling/importance_sampling_ratio/mean": 1.003359317779541, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.75310218334198, "clip_ratio/low_mean": 0.03947232058271766, "clip_ratio/low_min": 0.03947232058271766, "clip_ratio/high_mean": 0.07902801595628262, "clip_ratio/high_max": 0.07902801595628262, "clip_ratio/region_mean": 0.11850033653900027, "reward_total_mean": 0.9567533135414124, "reward_meter_mean": 0.987539529800415, "reward_meter_std": 0.020233290269970894, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9386054277420044, "reward_repeat_soft_std": 0.08637025207281113, "reward_judge_quality_mean": 0.8949999809265137, "reward_judge_quality_std": 0.0707106739282608, "reward_total_composite_mean": 0.9567533135414124, "reward_total_composite_std": 0.02130885049700737} {"timestamp_utc": "2026-04-13T01:15:07Z", "mode": "train", "global_step": 1191, "epoch": 0.11963837267704672, "loss": 0.0468, "grad_norm": 9.343894958496094, "learning_rate": 6.393939393939394e-06, "num_tokens": 2141402.0, "completions/mean_length": 80.625, "completions/min_length": 73.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.625, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9367615580558777, "rewards/meter/std": 0.15075694024562836, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8795038461685181, "rewards/repeat_soft/std": 0.08122512698173523, "rewards/judge_quality/mean": 0.3387500047683716, "rewards/judge_quality/std": 0.09538455307483673, "rewards/total_composite/mean": 0.7236180305480957, "rewards/total_composite/std": 0.06241123750805855, "reward": 0.7236180305480957, "reward_std": 0.06241123005747795, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10925004631280899, "sampling/sampling_logp_difference/max": 1.5458873510360718, "sampling/importance_sampling_ratio/min": 0.2131226658821106, "sampling/importance_sampling_ratio/mean": 1.0029542446136475, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7212669849395752, "clip_ratio/low_mean": 0.03583633853122592, "clip_ratio/low_min": 0.03583633853122592, "clip_ratio/high_mean": 0.053405508399009705, "clip_ratio/high_max": 0.053405508399009705, "clip_ratio/region_mean": 0.08924184693023562, "reward_total_mean": 0.7236180305480957, "reward_meter_mean": 0.9367615580558777, "reward_meter_std": 0.15075694024562836, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8795038461685181, "reward_repeat_soft_std": 0.08122512698173523, "reward_judge_quality_mean": 0.3387500047683716, "reward_judge_quality_std": 0.09538455307483673, "reward_total_composite_mean": 0.7236180305480957, "reward_total_composite_std": 0.06241123750805855} {"timestamp_utc": "2026-04-13T01:15:14Z", "mode": "train", "global_step": 1192, "epoch": 0.1197388247112004, "loss": -0.0209, "grad_norm": 15.69362735748291, "learning_rate": 6.390909090909091e-06, "num_tokens": 2142987.0, "completions/mean_length": 44.125, "completions/min_length": 33.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.125, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.41021284461021423, "rewards/meter/std": 0.41181138157844543, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9690903425216675, "rewards/repeat_soft/std": 0.023220641538500786, "rewards/judge_quality/mean": 0.4437499940395355, "rewards/judge_quality/std": 0.2084595113992691, "rewards/total_composite/mean": 0.5646297931671143, "rewards/total_composite/std": 0.18625620007514954, "reward": 0.5646297931671143, "reward_std": 0.18625618517398834, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16913826763629913, "sampling/sampling_logp_difference/max": 1.3471717834472656, "sampling/importance_sampling_ratio/min": 0.27241453528404236, "sampling/importance_sampling_ratio/mean": 1.029483437538147, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.315810389816761, "clip_ratio/low_mean": 0.09108951315283775, "clip_ratio/low_min": 0.09108951315283775, "clip_ratio/high_mean": 0.06947115436196327, "clip_ratio/high_max": 0.06947115436196327, "clip_ratio/region_mean": 0.16056066751480103, "reward_total_mean": 0.5646297931671143, "reward_meter_mean": 0.41021284461021423, "reward_meter_std": 0.41181138157844543, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9690903425216675, "reward_repeat_soft_std": 0.023220641538500786, "reward_judge_quality_mean": 0.4437499940395355, "reward_judge_quality_std": 0.2084595113992691, "reward_total_composite_mean": 0.5646297931671143, "reward_total_composite_std": 0.18625620007514954} {"timestamp_utc": "2026-04-13T01:15:21Z", "mode": "train", "global_step": 1193, "epoch": 0.11983927674535409, "loss": 0.0714, "grad_norm": 16.040164947509766, "learning_rate": 6.387878787878789e-06, "num_tokens": 2144653.0, "completions/mean_length": 34.25, "completions/min_length": 30.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.25, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.35983946919441223, "rewards/meter/std": 0.4452345371246338, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9699753522872925, "rewards/repeat_soft/std": 0.03587391600012779, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.4808002710342407, "rewards/total_composite/std": 0.20636042952537537, "reward": 0.4808002710342407, "reward_std": 0.20636042952537537, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12900127470493317, "sampling/sampling_logp_difference/max": 1.5122158527374268, "sampling/importance_sampling_ratio/min": 0.2484036535024643, "sampling/importance_sampling_ratio/mean": 1.0133907794952393, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6639175117015839, "clip_ratio/low_mean": 0.07774039870128036, "clip_ratio/low_min": 0.07774039870128036, "clip_ratio/high_mean": 0.018605169840157032, "clip_ratio/high_max": 0.018605169840157032, "clip_ratio/region_mean": 0.09634556854143739, "reward_total_mean": 0.4808002710342407, "reward_meter_mean": 0.35983946919441223, "reward_meter_std": 0.4452345371246338, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9699753522872925, "reward_repeat_soft_std": 0.03587391600012779, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.4808002710342407, "reward_total_composite_std": 0.20636042952537537} {"timestamp_utc": "2026-04-13T01:15:28Z", "mode": "train", "global_step": 1194, "epoch": 0.11993972877950779, "loss": 0.5569, "grad_norm": 17.63174819946289, "learning_rate": 6.384848484848485e-06, "num_tokens": 2145977.0, "completions/mean_length": 29.5, "completions/min_length": 18.0, "completions/max_length": 89.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.5, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 89.0, "rewards/meter/mean": 0.28216996788978577, "rewards/meter/std": 0.34351518750190735, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9329575896263123, "rewards/repeat_soft/std": 0.06946953386068344, "rewards/judge_quality/mean": 0.44875001907348633, "rewards/judge_quality/std": 0.21853001415729523, "rewards/total_composite/mean": 0.4696148633956909, "rewards/total_composite/std": 0.2681007981300354, "reward": 0.4696148633956909, "reward_std": 0.2681007981300354, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13718082010746002, "sampling/sampling_logp_difference/max": 1.4451370239257812, "sampling/importance_sampling_ratio/min": 0.23571377992630005, "sampling/importance_sampling_ratio/mean": 1.0267479419708252, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5249891132116318, "clip_ratio/low_mean": 0.04839324299246073, "clip_ratio/low_min": 0.04839324299246073, "clip_ratio/high_mean": 0.028906846418976784, "clip_ratio/high_max": 0.028906846418976784, "clip_ratio/region_mean": 0.07730008941143751, "reward_total_mean": 0.4696148633956909, "reward_meter_mean": 0.28216996788978577, "reward_meter_std": 0.34351518750190735, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9329575896263123, "reward_repeat_soft_std": 0.06946953386068344, "reward_judge_quality_mean": 0.44875001907348633, "reward_judge_quality_std": 0.21853001415729523, "reward_total_composite_mean": 0.4696148633956909, "reward_total_composite_std": 0.2681007981300354} {"timestamp_utc": "2026-04-13T01:15:40Z", "mode": "train", "global_step": 1195, "epoch": 0.12004018081366148, "loss": -0.0799, "grad_norm": 1.5044609308242798, "learning_rate": 6.381818181818182e-06, "num_tokens": 2147345.0, "completions/mean_length": 85.0, "completions/min_length": 22.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 24.000001907348633, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.7912251949310303, "rewards/meter/std": 0.34795448184013367, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9520467519760132, "rewards/repeat_soft/std": 0.026455143466591835, "rewards/judge_quality/mean": 0.3424999713897705, "rewards/judge_quality/std": 0.14606750011444092, "rewards/total_composite/mean": 0.6708810329437256, "rewards/total_composite/std": 0.2762202024459839, "reward": 0.6708810329437256, "reward_std": 0.2762202024459839, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16911520063877106, "sampling/sampling_logp_difference/max": 2.524245023727417, "sampling/importance_sampling_ratio/min": 0.08011877536773682, "sampling/importance_sampling_ratio/mean": 1.0050040483474731, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7946828342974186, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.13468413427472115, "clip_ratio/high_max": 0.13468413427472115, "clip_ratio/region_mean": 0.13468413427472115, "reward_total_mean": 0.6708810329437256, "reward_meter_mean": 0.7912251949310303, "reward_meter_std": 0.34795448184013367, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9520467519760132, "reward_repeat_soft_std": 0.026455143466591835, "reward_judge_quality_mean": 0.3424999713897705, "reward_judge_quality_std": 0.14606750011444092, "reward_total_composite_mean": 0.6708810329437256, "reward_total_composite_std": 0.2762202024459839} {"timestamp_utc": "2026-04-13T01:15:48Z", "mode": "train", "global_step": 1196, "epoch": 0.12014063284781516, "loss": 0.0509, "grad_norm": 10.319924354553223, "learning_rate": 6.37878787878788e-06, "num_tokens": 2149110.0, "completions/mean_length": 44.625, "completions/min_length": 37.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.625, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9201157093048096, "rewards/meter/std": 0.09424843639135361, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9568317532539368, "rewards/repeat_soft/std": 0.04574858024716377, "rewards/judge_quality/mean": 0.5349999666213989, "rewards/judge_quality/std": 0.1911618709564209, "rewards/total_composite/mean": 0.8202352523803711, "rewards/total_composite/std": 0.08694317936897278, "reward": 0.8202352523803711, "reward_std": 0.08694318681955338, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15460720658302307, "sampling/sampling_logp_difference/max": 1.3738155364990234, "sampling/importance_sampling_ratio/min": 0.2531392574310303, "sampling/importance_sampling_ratio/mean": 1.0126973390579224, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.166356772184372, "clip_ratio/low_mean": 0.07690286915749311, "clip_ratio/low_min": 0.07690286915749311, "clip_ratio/high_mean": 0.05163390561938286, "clip_ratio/high_max": 0.05163390561938286, "clip_ratio/region_mean": 0.12853677477687597, "reward_total_mean": 0.8202352523803711, "reward_meter_mean": 0.9201157093048096, "reward_meter_std": 0.09424843639135361, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9568317532539368, "reward_repeat_soft_std": 0.04574858024716377, "reward_judge_quality_mean": 0.5349999666213989, "reward_judge_quality_std": 0.1911618709564209, "reward_total_composite_mean": 0.8202352523803711, "reward_total_composite_std": 0.08694317936897278} {"timestamp_utc": "2026-04-13T01:15:54Z", "mode": "train", "global_step": 1197, "epoch": 0.12024108488196886, "loss": -0.0277, "grad_norm": 17.737882614135742, "learning_rate": 6.375757575757576e-06, "num_tokens": 2150813.0, "completions/mean_length": 37.875, "completions/min_length": 34.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.875, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.8493794798851013, "rewards/meter/std": 0.3422236144542694, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9020419716835022, "rewards/repeat_soft/std": 0.07071300595998764, "rewards/judge_quality/mean": 0.5049999952316284, "rewards/judge_quality/std": 0.16801361739635468, "rewards/total_composite/mean": 0.764549970626831, "rewards/total_composite/std": 0.19207435846328735, "reward": 0.764549970626831, "reward_std": 0.19207434356212616, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15707281231880188, "sampling/sampling_logp_difference/max": 2.8867950439453125, "sampling/importance_sampling_ratio/min": 0.05575462058186531, "sampling/importance_sampling_ratio/mean": 0.9969703555107117, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0832913964986801, "clip_ratio/low_mean": 0.02491830103099346, "clip_ratio/low_min": 0.02491830103099346, "clip_ratio/high_mean": 0.1104233362711966, "clip_ratio/high_max": 0.1104233362711966, "clip_ratio/region_mean": 0.13534163730219007, "reward_total_mean": 0.764549970626831, "reward_meter_mean": 0.8493794798851013, "reward_meter_std": 0.3422236144542694, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9020419716835022, "reward_repeat_soft_std": 0.07071300595998764, "reward_judge_quality_mean": 0.5049999952316284, "reward_judge_quality_std": 0.16801361739635468, "reward_total_composite_mean": 0.764549970626831, "reward_total_composite_std": 0.19207435846328735} {"timestamp_utc": "2026-04-13T01:16:06Z", "mode": "train", "global_step": 1198, "epoch": 0.12034153691612255, "loss": -0.1197, "grad_norm": 3.699859380722046, "learning_rate": 6.372727272727274e-06, "num_tokens": 2152632.0, "completions/mean_length": 117.375, "completions/min_length": 57.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 61.000003814697266, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.614952802658081, "rewards/meter/std": 0.42446842789649963, "rewards/count_adherence/mean": 0.65625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9734255075454712, "rewards/repeat_soft/std": 0.022068699821829796, "rewards/judge_quality/mean": 0.4150000214576721, "rewards/judge_quality/std": 0.18031719326972961, "rewards/total_composite/mean": 0.5826338529586792, "rewards/total_composite/std": 0.2896275222301483, "reward": 0.5826338529586792, "reward_std": 0.2896274924278259, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13495515286922455, "sampling/sampling_logp_difference/max": 1.2047042846679688, "sampling/importance_sampling_ratio/min": 0.29978063702583313, "sampling/importance_sampling_ratio/mean": 1.0264813899993896, "sampling/importance_sampling_ratio/max": 1.993887186050415, "entropy": 0.9120753481984138, "clip_ratio/low_mean": 0.04336701426655054, "clip_ratio/low_min": 0.04336701426655054, "clip_ratio/high_mean": 0.06471451185643673, "clip_ratio/high_max": 0.06471451185643673, "clip_ratio/region_mean": 0.10808152612298727, "reward_total_mean": 0.5826338529586792, "reward_meter_mean": 0.614952802658081, "reward_meter_std": 0.42446842789649963, "reward_count_adherence_mean": 0.65625, "reward_count_adherence_std": 0.2651650309562683, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9734255075454712, "reward_repeat_soft_std": 0.022068699821829796, "reward_judge_quality_mean": 0.4150000214576721, "reward_judge_quality_std": 0.18031719326972961, "reward_total_composite_mean": 0.5826338529586792, "reward_total_composite_std": 0.2896275222301483} {"timestamp_utc": "2026-04-13T01:16:12Z", "mode": "train", "global_step": 1199, "epoch": 0.12044198895027625, "loss": 0.0523, "grad_norm": 14.830349922180176, "learning_rate": 6.3696969696969706e-06, "num_tokens": 2154402.0, "completions/mean_length": 52.25, "completions/min_length": 49.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.25, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9218515753746033, "rewards/meter/std": 0.1992878019809723, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9778914451599121, "rewards/repeat_soft/std": 0.022302912548184395, "rewards/judge_quality/mean": 0.7987500429153442, "rewards/judge_quality/std": 0.22465451061725616, "rewards/total_composite/mean": 0.9022473096847534, "rewards/total_composite/std": 0.09912846237421036, "reward": 0.9022473096847534, "reward_std": 0.09912845492362976, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1336669623851776, "sampling/sampling_logp_difference/max": 1.357583999633789, "sampling/importance_sampling_ratio/min": 0.257281631231308, "sampling/importance_sampling_ratio/mean": 1.0220662355422974, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9519457593560219, "clip_ratio/low_mean": 0.047901757061481476, "clip_ratio/low_min": 0.047901757061481476, "clip_ratio/high_mean": 0.08612460363656282, "clip_ratio/high_max": 0.08612460363656282, "clip_ratio/region_mean": 0.1340263606980443, "reward_total_mean": 0.9022473096847534, "reward_meter_mean": 0.9218515753746033, "reward_meter_std": 0.1992878019809723, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9778914451599121, "reward_repeat_soft_std": 0.022302912548184395, "reward_judge_quality_mean": 0.7987500429153442, "reward_judge_quality_std": 0.22465451061725616, "reward_total_composite_mean": 0.9022473096847534, "reward_total_composite_std": 0.09912846237421036} {"timestamp_utc": "2026-04-13T01:16:19Z", "mode": "train", "global_step": 1200, "epoch": 0.12054244098442994, "loss": -0.0185, "grad_norm": 19.609376907348633, "learning_rate": 6.366666666666668e-06, "num_tokens": 2156028.0, "completions/mean_length": 43.25, "completions/min_length": 36.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.25, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.6932311654090881, "rewards/meter/std": 0.39263755083084106, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9616702795028687, "rewards/repeat_soft/std": 0.03556084632873535, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.6852460503578186, "rewards/total_composite/std": 0.17804981768131256, "reward": 0.6852460503578186, "reward_std": 0.17804981768131256, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15174373984336853, "sampling/sampling_logp_difference/max": 2.6058597564697266, "sampling/importance_sampling_ratio/min": 0.07383961975574493, "sampling/importance_sampling_ratio/mean": 1.0224682092666626, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9265835732221603, "clip_ratio/low_mean": 0.012896825559437275, "clip_ratio/low_min": 0.012896825559437275, "clip_ratio/high_mean": 0.09572415985167027, "clip_ratio/high_max": 0.09572415985167027, "clip_ratio/region_mean": 0.10862098541110754, "reward_total_mean": 0.6852460503578186, "reward_meter_mean": 0.6932311654090881, "reward_meter_std": 0.39263755083084106, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9616702795028687, "reward_repeat_soft_std": 0.03556084632873535, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.6852460503578186, "reward_total_composite_std": 0.17804981768131256} {"timestamp_utc": "2026-04-13T01:17:31Z", "mode": "eval", "global_step": 1200, "epoch": 0.12054244098442994, "eval_loss": NaN, "eval_runtime": 72.038, "eval_samples_per_second": 1.111, "eval_steps_per_second": 0.139, "eval_num_tokens": 2156028.0, "eval_completions/mean_length": 108.725, "eval_completions/min_length": 29.0, "eval_completions/max_length": 346.8, "eval_completions/clipped_ratio": 0.1125, "eval_completions/mean_terminated_length": 57.10500144958496, "eval_completions/min_terminated_length": 29.0, "eval_completions/max_terminated_length": 99.6, "eval_rewards/meter/mean": 0.6305708825588227, "eval_rewards/meter/std": 0.36522303223609925, "eval_rewards/count_adherence/mean": 0.7824999988079071, "eval_rewards/count_adherence/std": 0.2011160969734192, "eval_rewards/hard_gate/mean": 0.8875, "eval_rewards/hard_gate/std": 0.23946728110313414, "eval_rewards/repeat_soft/mean": 0.956658935546875, "eval_rewards/repeat_soft/std": 0.04269812703132629, "eval_rewards/judge_quality/mean": 0.382124999165535, "eval_rewards/judge_quality/std": 0.14539385680109262, "eval_rewards/total_composite/mean": 0.5833013236522675, "eval_rewards/total_composite/std": 0.24525272324681283, "eval_reward": 0.5833013236522675, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.08584631457924843, "eval_sampling/sampling_logp_difference/max": 0.976820421218872, "eval_sampling/importance_sampling_ratio/min": 0.387470480799675, "eval_sampling/importance_sampling_ratio/mean": 1.0240382313728333, "eval_sampling/importance_sampling_ratio/max": 1.624125564098358, "eval_entropy": 1.0200041711330414, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5833013236522675, "eval_reward_meter_mean": 0.6305708825588227, "eval_reward_meter_std": 0.36522303223609925, "eval_reward_count_adherence_mean": 0.7824999988079071, "eval_reward_count_adherence_std": 0.2011160969734192, "eval_reward_hard_gate_mean": 0.8875, "eval_reward_hard_gate_std": 0.23946728110313414, "eval_reward_repeat_soft_mean": 0.956658935546875, "eval_reward_repeat_soft_std": 0.04269812703132629, "eval_reward_judge_quality_mean": 0.382124999165535, "eval_reward_judge_quality_std": 0.14539385680109262, "eval_reward_total_composite_mean": 0.5833013236522675, "eval_reward_total_composite_std": 0.24525272324681283} {"timestamp_utc": "2026-04-13T01:17:41Z", "mode": "train", "global_step": 1201, "epoch": 0.12064289301858362, "loss": 0.0432, "grad_norm": 11.28005313873291, "learning_rate": 6.363636363636364e-06, "num_tokens": 2158396.0, "completions/mean_length": 86.0, "completions/min_length": 66.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.8183392286300659, "rewards/meter/std": 0.31433817744255066, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9526491165161133, "rewards/repeat_soft/std": 0.033176522701978683, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.13695022463798523, "rewards/total_composite/mean": 0.7256425619125366, "rewards/total_composite/std": 0.16777195036411285, "reward": 0.7256425619125366, "reward_std": 0.16777195036411285, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.131524458527565, "sampling/sampling_logp_difference/max": 1.58168625831604, "sampling/importance_sampling_ratio/min": 0.20562806725502014, "sampling/importance_sampling_ratio/mean": 0.9998653531074524, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.867162212729454, "clip_ratio/low_mean": 0.032543995417654514, "clip_ratio/low_min": 0.032543995417654514, "clip_ratio/high_mean": 0.11695955600589514, "clip_ratio/high_max": 0.11695955600589514, "clip_ratio/region_mean": 0.14950355142354965, "reward_total_mean": 0.7256425619125366, "reward_meter_mean": 0.8183392286300659, "reward_meter_std": 0.31433817744255066, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9526491165161133, "reward_repeat_soft_std": 0.033176522701978683, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.13695022463798523, "reward_total_composite_mean": 0.7256425619125366, "reward_total_composite_std": 0.16777195036411285} {"timestamp_utc": "2026-04-13T01:17:48Z", "mode": "train", "global_step": 1202, "epoch": 0.12074334505273732, "loss": 0.0345, "grad_norm": 36.32698440551758, "learning_rate": 6.3606060606060615e-06, "num_tokens": 2159787.0, "completions/mean_length": 20.875, "completions/min_length": 17.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.875, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.7616328001022339, "rewards/meter/std": 0.41072750091552734, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9536992311477661, "rewards/repeat_soft/std": 0.016395991668105125, "rewards/judge_quality/mean": 0.36374998092651367, "rewards/judge_quality/std": 0.09500939399003983, "rewards/total_composite/mean": 0.6972296237945557, "rewards/total_composite/std": 0.17409957945346832, "reward": 0.6972296237945557, "reward_std": 0.17409956455230713, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17035777866840363, "sampling/sampling_logp_difference/max": 1.2743911743164062, "sampling/importance_sampling_ratio/min": 0.27960115671157837, "sampling/importance_sampling_ratio/mean": 0.9961943030357361, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1897898316383362, "clip_ratio/low_mean": 0.033752859570086, "clip_ratio/low_min": 0.033752859570086, "clip_ratio/high_mean": 0.15033979713916779, "clip_ratio/high_max": 0.15033979713916779, "clip_ratio/region_mean": 0.1840926567092538, "reward_total_mean": 0.6972296237945557, "reward_meter_mean": 0.7616328001022339, "reward_meter_std": 0.41072750091552734, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9536992311477661, "reward_repeat_soft_std": 0.016395991668105125, "reward_judge_quality_mean": 0.36374998092651367, "reward_judge_quality_std": 0.09500939399003983, "reward_total_composite_mean": 0.6972296237945557, "reward_total_composite_std": 0.17409957945346832} {"timestamp_utc": "2026-04-13T01:18:00Z", "mode": "train", "global_step": 1203, "epoch": 0.12084379708689101, "loss": -0.0803, "grad_norm": 2.0878660678863525, "learning_rate": 6.357575757575758e-06, "num_tokens": 2161167.0, "completions/mean_length": 82.5, "completions/min_length": 18.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 21.142858505249023, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.7725033760070801, "rewards/meter/std": 0.3511694073677063, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9165509939193726, "rewards/repeat_soft/std": 0.08715125173330307, "rewards/judge_quality/mean": 0.3137499988079071, "rewards/judge_quality/std": 0.13845446705818176, "rewards/total_composite/mean": 0.6502816081047058, "rewards/total_composite/std": 0.27328795194625854, "reward": 0.6502816081047058, "reward_std": 0.27328792214393616, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12759530544281006, "sampling/sampling_logp_difference/max": 1.0411949157714844, "sampling/importance_sampling_ratio/min": 0.35303258895874023, "sampling/importance_sampling_ratio/mean": 1.0211232900619507, "sampling/importance_sampling_ratio/max": 1.8186278343200684, "entropy": 0.6787362396717072, "clip_ratio/low_mean": 0.013888888992369175, "clip_ratio/low_min": 0.013888888992369175, "clip_ratio/high_mean": 0.08824982354417443, "clip_ratio/high_max": 0.08824982354417443, "clip_ratio/region_mean": 0.10213871253654361, "reward_total_mean": 0.6502816081047058, "reward_meter_mean": 0.7725033760070801, "reward_meter_std": 0.3511694073677063, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9165509939193726, "reward_repeat_soft_std": 0.08715125173330307, "reward_judge_quality_mean": 0.3137499988079071, "reward_judge_quality_std": 0.13845446705818176, "reward_total_composite_mean": 0.6502816081047058, "reward_total_composite_std": 0.27328795194625854} {"timestamp_utc": "2026-04-13T01:18:12Z", "mode": "train", "global_step": 1204, "epoch": 0.1209442491210447, "loss": -0.1035, "grad_norm": 2.619563341140747, "learning_rate": 6.354545454545455e-06, "num_tokens": 2162687.0, "completions/mean_length": 173.0, "completions/min_length": 56.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 60.0, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.7282148003578186, "rewards/meter/std": 0.356042742729187, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9524650573730469, "rewards/repeat_soft/std": 0.046263907104730606, "rewards/judge_quality/mean": 0.32999998331069946, "rewards/judge_quality/std": 0.1697056144475937, "rewards/total_composite/mean": 0.5436998009681702, "rewards/total_composite/std": 0.28603559732437134, "reward": 0.5436998009681702, "reward_std": 0.28603559732437134, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15495428442955017, "sampling/sampling_logp_difference/max": 1.791762351989746, "sampling/importance_sampling_ratio/min": 0.1666661947965622, "sampling/importance_sampling_ratio/mean": 1.0396407842636108, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8256027474999428, "clip_ratio/low_mean": 0.01844262331724167, "clip_ratio/low_min": 0.01844262331724167, "clip_ratio/high_mean": 0.07712774630635977, "clip_ratio/high_max": 0.07712774630635977, "clip_ratio/region_mean": 0.09557036962360144, "reward_total_mean": 0.5436998009681702, "reward_meter_mean": 0.7282148003578186, "reward_meter_std": 0.356042742729187, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9524650573730469, "reward_repeat_soft_std": 0.046263907104730606, "reward_judge_quality_mean": 0.32999998331069946, "reward_judge_quality_std": 0.1697056144475937, "reward_total_composite_mean": 0.5436998009681702, "reward_total_composite_std": 0.28603559732437134} {"timestamp_utc": "2026-04-13T01:18:24Z", "mode": "train", "global_step": 1205, "epoch": 0.12104470115519839, "loss": -0.0745, "grad_norm": 2.0331175327301025, "learning_rate": 6.3515151515151516e-06, "num_tokens": 2164180.0, "completions/mean_length": 154.625, "completions/min_length": 31.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 35.5, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.5897265076637268, "rewards/meter/std": 0.4366273581981659, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9622437357902527, "rewards/repeat_soft/std": 0.041487570852041245, "rewards/judge_quality/mean": 0.4050000011920929, "rewards/judge_quality/std": 0.27401772141456604, "rewards/total_composite/mean": 0.5630647540092468, "rewards/total_composite/std": 0.37420472502708435, "reward": 0.5630647540092468, "reward_std": 0.37420472502708435, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16895942389965057, "sampling/sampling_logp_difference/max": 2.393033981323242, "sampling/importance_sampling_ratio/min": 0.09135209769010544, "sampling/importance_sampling_ratio/mean": 0.9879026412963867, "sampling/importance_sampling_ratio/max": 1.8965603113174438, "entropy": 0.5410313718020916, "clip_ratio/low_mean": 0.00937500037252903, "clip_ratio/low_min": 0.00937500037252903, "clip_ratio/high_mean": 0.08348816866055131, "clip_ratio/high_max": 0.08348816866055131, "clip_ratio/region_mean": 0.09286316903308034, "reward_total_mean": 0.5630647540092468, "reward_meter_mean": 0.5897265076637268, "reward_meter_std": 0.4366273581981659, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9622437357902527, "reward_repeat_soft_std": 0.041487570852041245, "reward_judge_quality_mean": 0.4050000011920929, "reward_judge_quality_std": 0.27401772141456604, "reward_total_composite_mean": 0.5630647540092468, "reward_total_composite_std": 0.37420472502708435} {"timestamp_utc": "2026-04-13T01:18:30Z", "mode": "train", "global_step": 1206, "epoch": 0.12114515318935208, "loss": 0.0555, "grad_norm": 13.28713607788086, "learning_rate": 6.34848484848485e-06, "num_tokens": 2166002.0, "completions/mean_length": 49.75, "completions/min_length": 42.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.75, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9896355867385864, "rewards/meter/std": 0.007893424481153488, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9058278203010559, "rewards/repeat_soft/std": 0.10548601299524307, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.8077937960624695, "rewards/total_composite/std": 0.01998145505785942, "reward": 0.8077937960624695, "reward_std": 0.019981462508440018, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15327255427837372, "sampling/sampling_logp_difference/max": 1.7133636474609375, "sampling/importance_sampling_ratio/min": 0.18025845289230347, "sampling/importance_sampling_ratio/mean": 1.005720615386963, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0464054271578789, "clip_ratio/low_mean": 0.04512896854430437, "clip_ratio/low_min": 0.04512896854430437, "clip_ratio/high_mean": 0.07281728507950902, "clip_ratio/high_max": 0.07281728507950902, "clip_ratio/region_mean": 0.11794625362381339, "reward_total_mean": 0.8077937960624695, "reward_meter_mean": 0.9896355867385864, "reward_meter_std": 0.007893424481153488, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9058278203010559, "reward_repeat_soft_std": 0.10548601299524307, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.8077937960624695, "reward_total_composite_std": 0.01998145505785942} {"timestamp_utc": "2026-04-13T01:18:37Z", "mode": "train", "global_step": 1207, "epoch": 0.12124560522350578, "loss": 0.0521, "grad_norm": 19.372974395751953, "learning_rate": 6.345454545454546e-06, "num_tokens": 2167560.0, "completions/mean_length": 27.75, "completions/min_length": 21.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.75, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.7669366598129272, "rewards/meter/std": 0.3658576011657715, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9195718765258789, "rewards/repeat_soft/std": 0.12141893804073334, "rewards/judge_quality/mean": 0.42499998211860657, "rewards/judge_quality/std": 0.0707106739282608, "rewards/total_composite/mean": 0.7145786285400391, "rewards/total_composite/std": 0.16011714935302734, "reward": 0.7145786285400391, "reward_std": 0.16011714935302734, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15736345946788788, "sampling/sampling_logp_difference/max": 1.292459487915039, "sampling/importance_sampling_ratio/min": 0.3299907147884369, "sampling/importance_sampling_ratio/mean": 1.0223535299301147, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1262102499604225, "clip_ratio/low_mean": 0.025457974523305893, "clip_ratio/low_min": 0.025457974523305893, "clip_ratio/high_mean": 0.1238472443073988, "clip_ratio/high_max": 0.1238472443073988, "clip_ratio/region_mean": 0.1493052188307047, "reward_total_mean": 0.7145786285400391, "reward_meter_mean": 0.7669366598129272, "reward_meter_std": 0.3658576011657715, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9195718765258789, "reward_repeat_soft_std": 0.12141893804073334, "reward_judge_quality_mean": 0.42499998211860657, "reward_judge_quality_std": 0.0707106739282608, "reward_total_composite_mean": 0.7145786285400391, "reward_total_composite_std": 0.16011714935302734} {"timestamp_utc": "2026-04-13T01:18:49Z", "mode": "train", "global_step": 1208, "epoch": 0.12134605725765947, "loss": -0.0948, "grad_norm": 3.4178733825683594, "learning_rate": 6.342424242424243e-06, "num_tokens": 2169274.0, "completions/mean_length": 99.25, "completions/min_length": 35.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 40.28571701049805, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.5287155508995056, "rewards/meter/std": 0.45853468775749207, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.1178511381149292, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9749672412872314, "rewards/repeat_soft/std": 0.026339536532759666, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.28965190052986145, "rewards/total_composite/mean": 0.5581613183021545, "rewards/total_composite/std": 0.26911816000938416, "reward": 0.5581613183021545, "reward_std": 0.26911813020706177, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15436972677707672, "sampling/sampling_logp_difference/max": 2.2280993461608887, "sampling/importance_sampling_ratio/min": 0.10773300379514694, "sampling/importance_sampling_ratio/mean": 1.0634591579437256, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.024716630578041, "clip_ratio/low_mean": 0.040785908699035645, "clip_ratio/low_min": 0.040785908699035645, "clip_ratio/high_mean": 0.06935883522965014, "clip_ratio/high_max": 0.06935883522965014, "clip_ratio/region_mean": 0.11014474392868578, "reward_total_mean": 0.5581613183021545, "reward_meter_mean": 0.5287155508995056, "reward_meter_std": 0.45853468775749207, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.1178511381149292, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9749672412872314, "reward_repeat_soft_std": 0.026339536532759666, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.28965190052986145, "reward_total_composite_mean": 0.5581613183021545, "reward_total_composite_std": 0.26911816000938416} {"timestamp_utc": "2026-04-13T01:19:06Z", "mode": "train", "global_step": 1209, "epoch": 0.12144650929181317, "loss": -0.1219, "grad_norm": 3.370405673980713, "learning_rate": 6.33939393939394e-06, "num_tokens": 2171081.0, "completions/mean_length": 113.875, "completions/min_length": 49.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 57.000003814697266, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.6493732333183289, "rewards/meter/std": 0.3579564392566681, "rewards/count_adherence/mean": 0.7250000238418579, "rewards/count_adherence/std": 0.2121320515871048, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.920163631439209, "rewards/repeat_soft/std": 0.05165758356451988, "rewards/judge_quality/mean": 0.4112499952316284, "rewards/judge_quality/std": 0.1797965168952942, "rewards/total_composite/mean": 0.5982338786125183, "rewards/total_composite/std": 0.2708955407142639, "reward": 0.5982338786125183, "reward_std": 0.2708955407142639, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1393721103668213, "sampling/sampling_logp_difference/max": 3.2147393226623535, "sampling/importance_sampling_ratio/min": 0.04016580060124397, "sampling/importance_sampling_ratio/mean": 0.9935009479522705, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6827701106667519, "clip_ratio/low_mean": 0.0309595400467515, "clip_ratio/low_min": 0.0309595400467515, "clip_ratio/high_mean": 0.09016166813671589, "clip_ratio/high_max": 0.09016166813671589, "clip_ratio/region_mean": 0.12112120818346739, "reward_total_mean": 0.5982338786125183, "reward_meter_mean": 0.6493732333183289, "reward_meter_std": 0.3579564392566681, "reward_count_adherence_mean": 0.7250000238418579, "reward_count_adherence_std": 0.2121320515871048, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.920163631439209, "reward_repeat_soft_std": 0.05165758356451988, "reward_judge_quality_mean": 0.4112499952316284, "reward_judge_quality_std": 0.1797965168952942, "reward_total_composite_mean": 0.5982338786125183, "reward_total_composite_std": 0.2708955407142639} {"timestamp_utc": "2026-04-13T01:19:13Z", "mode": "train", "global_step": 1210, "epoch": 0.12154696132596685, "loss": 0.0859, "grad_norm": 24.025903701782227, "learning_rate": 6.336363636363637e-06, "num_tokens": 2172415.0, "completions/mean_length": 20.75, "completions/min_length": 17.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.75, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.5901321172714233, "rewards/meter/std": 0.43426087498664856, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.6389344930648804, "rewards/total_composite/std": 0.19662253558635712, "reward": 0.6389344930648804, "reward_std": 0.1966225504875183, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20328354835510254, "sampling/sampling_logp_difference/max": 1.4163455963134766, "sampling/importance_sampling_ratio/min": 0.24259896576404572, "sampling/importance_sampling_ratio/mean": 1.0009359121322632, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0355529934167862, "clip_ratio/low_mean": 0.10748106054961681, "clip_ratio/low_min": 0.10748106054961681, "clip_ratio/high_mean": 0.11054566781967878, "clip_ratio/high_max": 0.11054566781967878, "clip_ratio/region_mean": 0.2180267283692956, "reward_total_mean": 0.6389344930648804, "reward_meter_mean": 0.5901321172714233, "reward_meter_std": 0.43426087498664856, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.6389344930648804, "reward_total_composite_std": 0.19662253558635712} {"timestamp_utc": "2026-04-13T01:19:20Z", "mode": "train", "global_step": 1211, "epoch": 0.12164741336012054, "loss": 0.0915, "grad_norm": 23.775766372680664, "learning_rate": 6.333333333333333e-06, "num_tokens": 2173898.0, "completions/mean_length": 32.375, "completions/min_length": 25.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.375, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.41448652744293213, "rewards/meter/std": 0.3312067687511444, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9897587895393372, "rewards/repeat_soft/std": 0.01303893979638815, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.5738698244094849, "rewards/total_composite/std": 0.16000837087631226, "reward": 0.5738698244094849, "reward_std": 0.16000838577747345, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17135204374790192, "sampling/sampling_logp_difference/max": 1.1227047443389893, "sampling/importance_sampling_ratio/min": 0.3253984749317169, "sampling/importance_sampling_ratio/mean": 1.0044654607772827, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.032541237771511, "clip_ratio/low_mean": 0.06469042925164104, "clip_ratio/low_min": 0.06469042925164104, "clip_ratio/high_mean": 0.06491054454818368, "clip_ratio/high_max": 0.06491054454818368, "clip_ratio/region_mean": 0.12960097379982471, "reward_total_mean": 0.5738698244094849, "reward_meter_mean": 0.41448652744293213, "reward_meter_std": 0.3312067687511444, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9897587895393372, "reward_repeat_soft_std": 0.01303893979638815, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.5738698244094849, "reward_total_composite_std": 0.16000837087631226} {"timestamp_utc": "2026-04-13T01:19:27Z", "mode": "train", "global_step": 1212, "epoch": 0.12174786539427424, "loss": 0.007, "grad_norm": 8.413917541503906, "learning_rate": 6.330303030303031e-06, "num_tokens": 2176187.0, "completions/mean_length": 92.125, "completions/min_length": 87.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.125, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9670178890228271, "rewards/meter/std": 0.053708214312791824, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8116468191146851, "rewards/repeat_soft/std": 0.06922277808189392, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.781072735786438, "rewards/total_composite/std": 0.059025272727012634, "reward": 0.781072735786438, "reward_std": 0.05902528762817383, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12528035044670105, "sampling/sampling_logp_difference/max": 1.6174979209899902, "sampling/importance_sampling_ratio/min": 0.19839447736740112, "sampling/importance_sampling_ratio/mean": 1.0265941619873047, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0307793989777565, "clip_ratio/low_mean": 0.08120123855769634, "clip_ratio/low_min": 0.08120123855769634, "clip_ratio/high_mean": 0.03271960932761431, "clip_ratio/high_max": 0.03271960932761431, "clip_ratio/region_mean": 0.11392084788531065, "reward_total_mean": 0.781072735786438, "reward_meter_mean": 0.9670178890228271, "reward_meter_std": 0.053708214312791824, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8116468191146851, "reward_repeat_soft_std": 0.06922277808189392, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.781072735786438, "reward_total_composite_std": 0.059025272727012634} {"timestamp_utc": "2026-04-13T01:19:33Z", "mode": "train", "global_step": 1213, "epoch": 0.12184831742842793, "loss": -0.0804, "grad_norm": 13.858150482177734, "learning_rate": 6.327272727272727e-06, "num_tokens": 2177839.0, "completions/mean_length": 40.5, "completions/min_length": 33.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.5, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.9739177227020264, "rewards/meter/std": 0.015471726655960083, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.981476902961731, "rewards/repeat_soft/std": 0.012882005423307419, "rewards/judge_quality/mean": 0.47999998927116394, "rewards/judge_quality/std": 0.19071295857429504, "rewards/total_composite/mean": 0.8304107189178467, "rewards/total_composite/std": 0.06085222214460373, "reward": 0.8304107189178467, "reward_std": 0.060852229595184326, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1684126853942871, "sampling/sampling_logp_difference/max": 1.2159833908081055, "sampling/importance_sampling_ratio/min": 0.29641836881637573, "sampling/importance_sampling_ratio/mean": 1.0315505266189575, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3935264199972153, "clip_ratio/low_mean": 0.13688214495778084, "clip_ratio/low_min": 0.13688214495778084, "clip_ratio/high_mean": 0.029999999329447746, "clip_ratio/high_max": 0.029999999329447746, "clip_ratio/region_mean": 0.16688214428722858, "reward_total_mean": 0.8304107189178467, "reward_meter_mean": 0.9739177227020264, "reward_meter_std": 0.015471726655960083, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.981476902961731, "reward_repeat_soft_std": 0.012882005423307419, "reward_judge_quality_mean": 0.47999998927116394, "reward_judge_quality_std": 0.19071295857429504, "reward_total_composite_mean": 0.8304107189178467, "reward_total_composite_std": 0.06085222214460373} {"timestamp_utc": "2026-04-13T01:19:41Z", "mode": "train", "global_step": 1214, "epoch": 0.12194876946258161, "loss": 0.0397, "grad_norm": 20.62298011779785, "learning_rate": 6.324242424242425e-06, "num_tokens": 2179437.0, "completions/mean_length": 34.75, "completions/min_length": 32.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.75, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.7509653568267822, "rewards/meter/std": 0.3258548080921173, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9791094064712524, "rewards/repeat_soft/std": 0.01996542513370514, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7140953540802002, "rewards/total_composite/std": 0.14972354471683502, "reward": 0.7140953540802002, "reward_std": 0.14972354471683502, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1283295452594757, "sampling/sampling_logp_difference/max": 1.3552603721618652, "sampling/importance_sampling_ratio/min": 0.25788015127182007, "sampling/importance_sampling_ratio/mean": 1.0195896625518799, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7292398139834404, "clip_ratio/low_mean": 0.02552521089091897, "clip_ratio/low_min": 0.02552521089091897, "clip_ratio/high_mean": 0.11317528923973441, "clip_ratio/high_max": 0.11317528923973441, "clip_ratio/region_mean": 0.13870050013065338, "reward_total_mean": 0.7140953540802002, "reward_meter_mean": 0.7509653568267822, "reward_meter_std": 0.3258548080921173, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9791094064712524, "reward_repeat_soft_std": 0.01996542513370514, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7140953540802002, "reward_total_composite_std": 0.14972354471683502} {"timestamp_utc": "2026-04-13T01:19:53Z", "mode": "train", "global_step": 1215, "epoch": 0.1220492214967353, "loss": -0.1532, "grad_norm": 2.6866612434387207, "learning_rate": 6.3212121212121216e-06, "num_tokens": 2181405.0, "completions/mean_length": 193.0, "completions/min_length": 80.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 86.66667175292969, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.35920554399490356, "rewards/meter/std": 0.2832556664943695, "rewards/count_adherence/mean": 0.675000011920929, "rewards/count_adherence/std": 0.23754701018333435, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.885715126991272, "rewards/repeat_soft/std": 0.08996167778968811, "rewards/judge_quality/mean": 0.32749998569488525, "rewards/judge_quality/std": 0.17127670347690582, "rewards/total_composite/mean": 0.42599695920944214, "rewards/total_composite/std": 0.2310992032289505, "reward": 0.42599695920944214, "reward_std": 0.2310991734266281, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13031448423862457, "sampling/sampling_logp_difference/max": 2.2741992473602295, "sampling/importance_sampling_ratio/min": 0.10287925601005554, "sampling/importance_sampling_ratio/mean": 1.0245901346206665, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5356778465211391, "clip_ratio/low_mean": 0.0029761905316263437, "clip_ratio/low_min": 0.0029761905316263437, "clip_ratio/high_mean": 0.05996053013950586, "clip_ratio/high_max": 0.05996053013950586, "clip_ratio/region_mean": 0.0629367206711322, "reward_total_mean": 0.42599695920944214, "reward_meter_mean": 0.35920554399490356, "reward_meter_std": 0.2832556664943695, "reward_count_adherence_mean": 0.675000011920929, "reward_count_adherence_std": 0.23754701018333435, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.885715126991272, "reward_repeat_soft_std": 0.08996167778968811, "reward_judge_quality_mean": 0.32749998569488525, "reward_judge_quality_std": 0.17127670347690582, "reward_total_composite_mean": 0.42599695920944214, "reward_total_composite_std": 0.2310992032289505} {"timestamp_utc": "2026-04-13T01:20:01Z", "mode": "train", "global_step": 1216, "epoch": 0.122149673530889, "loss": 0.0485, "grad_norm": 24.08833122253418, "learning_rate": 6.318181818181819e-06, "num_tokens": 2182946.0, "completions/mean_length": 40.625, "completions/min_length": 16.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.625, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.8792199492454529, "rewards/meter/std": 0.28943711519241333, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9751440286636353, "rewards/repeat_soft/std": 0.01947917975485325, "rewards/judge_quality/mean": 0.4437499940395355, "rewards/judge_quality/std": 0.13845448195934296, "rewards/total_composite/mean": 0.71722412109375, "rewards/total_composite/std": 0.2941586673259735, "reward": 0.71722412109375, "reward_std": 0.2941586673259735, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14839661121368408, "sampling/sampling_logp_difference/max": 1.421335220336914, "sampling/importance_sampling_ratio/min": 0.24139149487018585, "sampling/importance_sampling_ratio/mean": 1.0301047563552856, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.078417383134365, "clip_ratio/low_mean": 0.014204545877873898, "clip_ratio/low_min": 0.014204545877873898, "clip_ratio/high_mean": 0.1298798080533743, "clip_ratio/high_max": 0.1298798080533743, "clip_ratio/region_mean": 0.1440843539312482, "reward_total_mean": 0.71722412109375, "reward_meter_mean": 0.8792199492454529, "reward_meter_std": 0.28943711519241333, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9751440286636353, "reward_repeat_soft_std": 0.01947917975485325, "reward_judge_quality_mean": 0.4437499940395355, "reward_judge_quality_std": 0.13845448195934296, "reward_total_composite_mean": 0.71722412109375, "reward_total_composite_std": 0.2941586673259735} {"timestamp_utc": "2026-04-13T01:20:07Z", "mode": "train", "global_step": 1217, "epoch": 0.1222501255650427, "loss": -0.0551, "grad_norm": 14.4071683883667, "learning_rate": 6.315151515151515e-06, "num_tokens": 2184432.0, "completions/mean_length": 43.75, "completions/min_length": 25.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.75, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.6294839978218079, "rewards/meter/std": 0.35907459259033203, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9736183285713196, "rewards/repeat_soft/std": 0.03277510777115822, "rewards/judge_quality/mean": 0.48875001072883606, "rewards/judge_quality/std": 0.15797266364097595, "rewards/total_composite/mean": 0.5119444131851196, "rewards/total_composite/std": 0.34176960587501526, "reward": 0.5119444131851196, "reward_std": 0.34176960587501526, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18527576327323914, "sampling/sampling_logp_difference/max": 1.526686429977417, "sampling/importance_sampling_ratio/min": 0.2172543853521347, "sampling/importance_sampling_ratio/mean": 1.0077873468399048, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.215646043419838, "clip_ratio/low_mean": 0.04519696347415447, "clip_ratio/low_min": 0.04519696347415447, "clip_ratio/high_mean": 0.10383122134953737, "clip_ratio/high_max": 0.10383122134953737, "clip_ratio/region_mean": 0.14902818482369184, "reward_total_mean": 0.5119444131851196, "reward_meter_mean": 0.6294839978218079, "reward_meter_std": 0.35907459259033203, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9736183285713196, "reward_repeat_soft_std": 0.03277510777115822, "reward_judge_quality_mean": 0.48875001072883606, "reward_judge_quality_std": 0.15797266364097595, "reward_total_composite_mean": 0.5119444131851196, "reward_total_composite_std": 0.34176960587501526} {"timestamp_utc": "2026-04-13T01:20:14Z", "mode": "train", "global_step": 1218, "epoch": 0.12235057759919639, "loss": 0.0151, "grad_norm": 13.361787796020508, "learning_rate": 6.3121212121212125e-06, "num_tokens": 2186187.0, "completions/mean_length": 61.375, "completions/min_length": 56.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.375, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.987122654914856, "rewards/meter/std": 0.009147382341325283, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9510934352874756, "rewards/repeat_soft/std": 0.05044485256075859, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.6922432780265808, "rewards/total_composite/std": 0.2816990911960602, "reward": 0.6922432780265808, "reward_std": 0.2816990613937378, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1283133327960968, "sampling/sampling_logp_difference/max": 2.8390626907348633, "sampling/importance_sampling_ratio/min": 0.05848045274615288, "sampling/importance_sampling_ratio/mean": 1.022992730140686, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8650243431329727, "clip_ratio/low_mean": 0.01844262331724167, "clip_ratio/low_min": 0.01844262331724167, "clip_ratio/high_mean": 0.09830616787075996, "clip_ratio/high_max": 0.09830616787075996, "clip_ratio/region_mean": 0.11674879118800163, "reward_total_mean": 0.6922432780265808, "reward_meter_mean": 0.987122654914856, "reward_meter_std": 0.009147382341325283, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9510934352874756, "reward_repeat_soft_std": 0.05044485256075859, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.6922432780265808, "reward_total_composite_std": 0.2816990911960602} {"timestamp_utc": "2026-04-13T01:20:21Z", "mode": "train", "global_step": 1219, "epoch": 0.12245102963335007, "loss": 0.0075, "grad_norm": 14.491744041442871, "learning_rate": 6.309090909090909e-06, "num_tokens": 2187864.0, "completions/mean_length": 49.625, "completions/min_length": 45.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.625, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.26350611448287964, "rewards/meter/std": 0.3074663281440735, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9804191589355469, "rewards/repeat_soft/std": 0.011817132122814655, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.44874468445777893, "rewards/total_composite/std": 0.13403671979904175, "reward": 0.44874468445777893, "reward_std": 0.13403670489788055, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17139670252799988, "sampling/sampling_logp_difference/max": 2.750115394592285, "sampling/importance_sampling_ratio/min": 0.06392048299312592, "sampling/importance_sampling_ratio/mean": 1.015247106552124, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0021399036049843, "clip_ratio/low_mean": 0.07014203257858753, "clip_ratio/low_min": 0.07014203257858753, "clip_ratio/high_mean": 0.0688997833058238, "clip_ratio/high_max": 0.0688997833058238, "clip_ratio/region_mean": 0.13904181588441133, "reward_total_mean": 0.44874468445777893, "reward_meter_mean": 0.26350611448287964, "reward_meter_std": 0.3074663281440735, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9804191589355469, "reward_repeat_soft_std": 0.011817132122814655, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.44874468445777893, "reward_total_composite_std": 0.13403671979904175} {"timestamp_utc": "2026-04-13T01:20:28Z", "mode": "train", "global_step": 1220, "epoch": 0.12255148166750376, "loss": -0.0046, "grad_norm": 16.07826805114746, "learning_rate": 6.306060606060607e-06, "num_tokens": 2189720.0, "completions/mean_length": 59.0, "completions/min_length": 53.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.0, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.799335241317749, "rewards/meter/std": 0.3423990309238434, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9095147252082825, "rewards/repeat_soft/std": 0.08442886918783188, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.683902382850647, "rewards/total_composite/std": 0.17310036718845367, "reward": 0.683902382850647, "reward_std": 0.17310035228729248, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12967722117900848, "sampling/sampling_logp_difference/max": 1.3078465461730957, "sampling/importance_sampling_ratio/min": 0.2704017460346222, "sampling/importance_sampling_ratio/mean": 1.001846194267273, "sampling/importance_sampling_ratio/max": 1.837649941444397, "entropy": 0.8761199563741684, "clip_ratio/low_mean": 0.02282681968063116, "clip_ratio/low_min": 0.02282681968063116, "clip_ratio/high_mean": 0.10147323831915855, "clip_ratio/high_max": 0.10147323831915855, "clip_ratio/region_mean": 0.12430005799978971, "reward_total_mean": 0.683902382850647, "reward_meter_mean": 0.799335241317749, "reward_meter_std": 0.3423990309238434, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9095147252082825, "reward_repeat_soft_std": 0.08442886918783188, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.683902382850647, "reward_total_composite_std": 0.17310036718845367} {"timestamp_utc": "2026-04-13T01:20:35Z", "mode": "train", "global_step": 1221, "epoch": 0.12265193370165746, "loss": -0.0131, "grad_norm": 13.46181869506836, "learning_rate": 6.303030303030303e-06, "num_tokens": 2191333.0, "completions/mean_length": 47.625, "completions/min_length": 44.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.625, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9836236238479614, "rewards/meter/std": 0.016504371538758278, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9865393042564392, "rewards/repeat_soft/std": 0.018572963774204254, "rewards/judge_quality/mean": 0.7687499523162842, "rewards/judge_quality/std": 0.2228027880191803, "rewards/total_composite/mean": 0.921909511089325, "rewards/total_composite/std": 0.07020371407270432, "reward": 0.921909511089325, "reward_std": 0.07020371407270432, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11231357604265213, "sampling/sampling_logp_difference/max": 1.58522367477417, "sampling/importance_sampling_ratio/min": 0.20490196347236633, "sampling/importance_sampling_ratio/mean": 0.9892477989196777, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8268293365836143, "clip_ratio/low_mean": 0.04819832928478718, "clip_ratio/low_min": 0.04819832928478718, "clip_ratio/high_mean": 0.08876867406070232, "clip_ratio/high_max": 0.08876867406070232, "clip_ratio/region_mean": 0.1369670033454895, "reward_total_mean": 0.921909511089325, "reward_meter_mean": 0.9836236238479614, "reward_meter_std": 0.016504371538758278, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9865393042564392, "reward_repeat_soft_std": 0.018572963774204254, "reward_judge_quality_mean": 0.7687499523162842, "reward_judge_quality_std": 0.2228027880191803, "reward_total_composite_mean": 0.921909511089325, "reward_total_composite_std": 0.07020371407270432} {"timestamp_utc": "2026-04-13T01:20:42Z", "mode": "train", "global_step": 1222, "epoch": 0.12275238573581115, "loss": 0.0615, "grad_norm": 13.936372756958008, "learning_rate": 6.300000000000001e-06, "num_tokens": 2192865.0, "completions/mean_length": 41.5, "completions/min_length": 39.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.5, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.569190502166748, "rewards/meter/std": 0.38121703267097473, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9844129085540771, "rewards/repeat_soft/std": 0.015649747103452682, "rewards/judge_quality/mean": 0.5137500166893005, "rewards/judge_quality/std": 0.13741882145404816, "rewards/total_composite/mean": 0.6087020039558411, "rewards/total_composite/std": 0.18504807353019714, "reward": 0.6087020039558411, "reward_std": 0.18504807353019714, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14609406888484955, "sampling/sampling_logp_difference/max": 1.4259862899780273, "sampling/importance_sampling_ratio/min": 0.24027137458324432, "sampling/importance_sampling_ratio/mean": 1.0093727111816406, "sampling/importance_sampling_ratio/max": 1.9626421928405762, "entropy": 0.8918471932411194, "clip_ratio/low_mean": 0.06017287261784077, "clip_ratio/low_min": 0.06017287261784077, "clip_ratio/high_mean": 0.09797129058279097, "clip_ratio/high_max": 0.09797129058279097, "clip_ratio/region_mean": 0.15814416320063174, "reward_total_mean": 0.6087020039558411, "reward_meter_mean": 0.569190502166748, "reward_meter_std": 0.38121703267097473, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9844129085540771, "reward_repeat_soft_std": 0.015649747103452682, "reward_judge_quality_mean": 0.5137500166893005, "reward_judge_quality_std": 0.13741882145404816, "reward_total_composite_mean": 0.6087020039558411, "reward_total_composite_std": 0.18504807353019714} {"timestamp_utc": "2026-04-13T01:20:49Z", "mode": "train", "global_step": 1223, "epoch": 0.12285283776996485, "loss": 0.0585, "grad_norm": 16.54353141784668, "learning_rate": 6.296969696969697e-06, "num_tokens": 2194789.0, "completions/mean_length": 66.5, "completions/min_length": 62.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.755329966545105, "rewards/meter/std": 0.31992125511169434, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9675406813621521, "rewards/repeat_soft/std": 0.014440210536122322, "rewards/judge_quality/mean": 0.4899999797344208, "rewards/judge_quality/std": 0.11501555144786835, "rewards/total_composite/mean": 0.6961525678634644, "rewards/total_composite/std": 0.13243772089481354, "reward": 0.6961525678634644, "reward_std": 0.13243772089481354, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1496732085943222, "sampling/sampling_logp_difference/max": 2.519742965698242, "sampling/importance_sampling_ratio/min": 0.08048029243946075, "sampling/importance_sampling_ratio/mean": 0.9935014843940735, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8209086060523987, "clip_ratio/low_mean": 0.04324495047330856, "clip_ratio/low_min": 0.04324495047330856, "clip_ratio/high_mean": 0.08651740569621325, "clip_ratio/high_max": 0.08651740569621325, "clip_ratio/region_mean": 0.1297623561695218, "reward_total_mean": 0.6961525678634644, "reward_meter_mean": 0.755329966545105, "reward_meter_std": 0.31992125511169434, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9675406813621521, "reward_repeat_soft_std": 0.014440210536122322, "reward_judge_quality_mean": 0.4899999797344208, "reward_judge_quality_std": 0.11501555144786835, "reward_total_composite_mean": 0.6961525678634644, "reward_total_composite_std": 0.13243772089481354} {"timestamp_utc": "2026-04-13T01:20:56Z", "mode": "train", "global_step": 1224, "epoch": 0.12295328980411853, "loss": 0.0091, "grad_norm": 10.113824844360352, "learning_rate": 6.293939393939394e-06, "num_tokens": 2196880.0, "completions/mean_length": 74.375, "completions/min_length": 69.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.375, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9843785166740417, "rewards/meter/std": 0.002666675951331854, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9252215623855591, "rewards/repeat_soft/std": 0.05883277207612991, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7814924716949463, "rewards/total_composite/std": 0.006550771649926901, "reward": 0.7814924716949463, "reward_std": 0.006550775840878487, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14991052448749542, "sampling/sampling_logp_difference/max": 1.3727607727050781, "sampling/importance_sampling_ratio/min": 0.2534063756465912, "sampling/importance_sampling_ratio/mean": 1.0296634435653687, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2089199870824814, "clip_ratio/low_mean": 0.04496111534535885, "clip_ratio/low_min": 0.04496111534535885, "clip_ratio/high_mean": 0.08841325528919697, "clip_ratio/high_max": 0.08841325528919697, "clip_ratio/region_mean": 0.13337437063455582, "reward_total_mean": 0.7814924716949463, "reward_meter_mean": 0.9843785166740417, "reward_meter_std": 0.002666675951331854, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9252215623855591, "reward_repeat_soft_std": 0.05883277207612991, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7814924716949463, "reward_total_composite_std": 0.006550771649926901} {"timestamp_utc": "2026-04-13T01:21:03Z", "mode": "train", "global_step": 1225, "epoch": 0.12305374183827222, "loss": -0.0083, "grad_norm": 13.000123023986816, "learning_rate": 6.290909090909092e-06, "num_tokens": 2198380.0, "completions/mean_length": 37.5, "completions/min_length": 31.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.5, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9129583239555359, "rewards/meter/std": 0.1999291479587555, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9837520122528076, "rewards/repeat_soft/std": 0.011667110025882721, "rewards/judge_quality/mean": 0.71875, "rewards/judge_quality/std": 0.23805388808250427, "rewards/total_composite/mean": 0.8748314380645752, "rewards/total_composite/std": 0.13699787855148315, "reward": 0.8748314380645752, "reward_std": 0.13699787855148315, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11805222183465958, "sampling/sampling_logp_difference/max": 1.1654033660888672, "sampling/importance_sampling_ratio/min": 0.3117968440055847, "sampling/importance_sampling_ratio/mean": 1.0152842998504639, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.803429052233696, "clip_ratio/low_mean": 0.04605310899205506, "clip_ratio/low_min": 0.04605310899205506, "clip_ratio/high_mean": 0.0856136791408062, "clip_ratio/high_max": 0.0856136791408062, "clip_ratio/region_mean": 0.13166678813286126, "reward_total_mean": 0.8748314380645752, "reward_meter_mean": 0.9129583239555359, "reward_meter_std": 0.1999291479587555, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9837520122528076, "reward_repeat_soft_std": 0.011667110025882721, "reward_judge_quality_mean": 0.71875, "reward_judge_quality_std": 0.23805388808250427, "reward_total_composite_mean": 0.8748314380645752, "reward_total_composite_std": 0.13699787855148315} {"timestamp_utc": "2026-04-13T01:21:10Z", "mode": "train", "global_step": 1226, "epoch": 0.12315419387242592, "loss": 0.0558, "grad_norm": 12.698612213134766, "learning_rate": 6.287878787878788e-06, "num_tokens": 2199977.0, "completions/mean_length": 40.625, "completions/min_length": 37.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.625, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.9800806045532227, "rewards/meter/std": 0.02270977757871151, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9873234629631042, "rewards/repeat_soft/std": 0.012976191006600857, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720350325107574, "rewards/total_composite/mean": 0.855518639087677, "rewards/total_composite/std": 0.07061190158128738, "reward": 0.855518639087677, "reward_std": 0.07061190158128738, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14618054032325745, "sampling/sampling_logp_difference/max": 1.2328052520751953, "sampling/importance_sampling_ratio/min": 0.30333074927330017, "sampling/importance_sampling_ratio/mean": 1.0490299463272095, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0132408812642097, "clip_ratio/low_mean": 0.1114133857190609, "clip_ratio/low_min": 0.1114133857190609, "clip_ratio/high_mean": 0.02626689150929451, "clip_ratio/high_max": 0.02626689150929451, "clip_ratio/region_mean": 0.1376802772283554, "reward_total_mean": 0.855518639087677, "reward_meter_mean": 0.9800806045532227, "reward_meter_std": 0.02270977757871151, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9873234629631042, "reward_repeat_soft_std": 0.012976191006600857, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720350325107574, "reward_total_composite_mean": 0.855518639087677, "reward_total_composite_std": 0.07061190158128738} {"timestamp_utc": "2026-04-13T01:21:16Z", "mode": "train", "global_step": 1227, "epoch": 0.12325464590657961, "loss": -0.0353, "grad_norm": 25.43566131591797, "learning_rate": 6.284848484848486e-06, "num_tokens": 2201343.0, "completions/mean_length": 19.75, "completions/min_length": 16.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.75, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.8953582048416138, "rewards/meter/std": 0.21190501749515533, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9575770497322083, "rewards/repeat_soft/std": 0.007982193492352962, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.776918888092041, "rewards/total_composite/std": 0.09649389982223511, "reward": 0.776918888092041, "reward_std": 0.09649389982223511, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18038514256477356, "sampling/sampling_logp_difference/max": 1.289259433746338, "sampling/importance_sampling_ratio/min": 0.2754747271537781, "sampling/importance_sampling_ratio/mean": 1.040266752243042, "sampling/importance_sampling_ratio/max": 1.9378445148468018, "entropy": 1.0907242894172668, "clip_ratio/low_mean": 0.029513888992369175, "clip_ratio/low_min": 0.029513888992369175, "clip_ratio/high_mean": 0.08537783846259117, "clip_ratio/high_max": 0.08537783846259117, "clip_ratio/region_mean": 0.11489172745496035, "reward_total_mean": 0.776918888092041, "reward_meter_mean": 0.8953582048416138, "reward_meter_std": 0.21190501749515533, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9575770497322083, "reward_repeat_soft_std": 0.007982193492352962, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.776918888092041, "reward_total_composite_std": 0.09649389982223511} {"timestamp_utc": "2026-04-13T01:21:27Z", "mode": "train", "global_step": 1228, "epoch": 0.1233550979407333, "loss": -0.1149, "grad_norm": 3.28853440284729, "learning_rate": 6.2818181818181825e-06, "num_tokens": 2202889.0, "completions/mean_length": 101.25, "completions/min_length": 38.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 42.57143020629883, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.7472759485244751, "rewards/meter/std": 0.37886419892311096, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9859052896499634, "rewards/repeat_soft/std": 0.014757425524294376, "rewards/judge_quality/mean": 0.47749999165534973, "rewards/judge_quality/std": 0.2589401602745056, "rewards/total_composite/mean": 0.7187396883964539, "rewards/total_composite/std": 0.2426249384880066, "reward": 0.7187396883964539, "reward_std": 0.2426249384880066, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17791730165481567, "sampling/sampling_logp_difference/max": 1.8908085823059082, "sampling/importance_sampling_ratio/min": 0.15094970166683197, "sampling/importance_sampling_ratio/mean": 1.0125882625579834, "sampling/importance_sampling_ratio/max": 1.8963435888290405, "entropy": 1.0427603125572205, "clip_ratio/low_mean": 0.009868420660495758, "clip_ratio/low_min": 0.009868420660495758, "clip_ratio/high_mean": 0.13902809284627438, "clip_ratio/high_max": 0.13902809284627438, "clip_ratio/region_mean": 0.14889651350677013, "reward_total_mean": 0.7187396883964539, "reward_meter_mean": 0.7472759485244751, "reward_meter_std": 0.37886419892311096, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9859052896499634, "reward_repeat_soft_std": 0.014757425524294376, "reward_judge_quality_mean": 0.47749999165534973, "reward_judge_quality_std": 0.2589401602745056, "reward_total_composite_mean": 0.7187396883964539, "reward_total_composite_std": 0.2426249384880066} {"timestamp_utc": "2026-04-13T01:21:33Z", "mode": "train", "global_step": 1229, "epoch": 0.12345554997488699, "loss": 0.0389, "grad_norm": 31.764644622802734, "learning_rate": 6.27878787878788e-06, "num_tokens": 2204315.0, "completions/mean_length": 31.25, "completions/min_length": 25.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.25, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.7388919591903687, "rewards/meter/std": 0.32252785563468933, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9965760707855225, "rewards/repeat_soft/std": 0.004902609623968601, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.7854089736938477, "rewards/total_composite/std": 0.16331852972507477, "reward": 0.7854089736938477, "reward_std": 0.16331851482391357, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1293390542268753, "sampling/sampling_logp_difference/max": 1.2150169610977173, "sampling/importance_sampling_ratio/min": 0.2967049777507782, "sampling/importance_sampling_ratio/mean": 1.0395256280899048, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7214206606149673, "clip_ratio/low_mean": 0.07418168801814318, "clip_ratio/low_min": 0.07418168801814318, "clip_ratio/high_mean": 0.05147569486871362, "clip_ratio/high_max": 0.05147569486871362, "clip_ratio/region_mean": 0.1256573828868568, "reward_total_mean": 0.7854089736938477, "reward_meter_mean": 0.7388919591903687, "reward_meter_std": 0.32252785563468933, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9965760707855225, "reward_repeat_soft_std": 0.004902609623968601, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.7854089736938477, "reward_total_composite_std": 0.16331852972507477} {"timestamp_utc": "2026-04-13T01:21:41Z", "mode": "train", "global_step": 1230, "epoch": 0.12355600200904068, "loss": 0.0193, "grad_norm": 8.704459190368652, "learning_rate": 6.275757575757576e-06, "num_tokens": 2206936.0, "completions/mean_length": 128.625, "completions/min_length": 123.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.625, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.9869276285171509, "rewards/meter/std": 0.016702743247151375, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9433026313781738, "rewards/repeat_soft/std": 0.023325536400079727, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7830727100372314, "rewards/total_composite/std": 0.017859673127532005, "reward": 0.7830727100372314, "reward_std": 0.01785965822637081, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13517242670059204, "sampling/sampling_logp_difference/max": 3.2558844089508057, "sampling/importance_sampling_ratio/min": 0.038546714931726456, "sampling/importance_sampling_ratio/mean": 1.0110551118850708, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9993724897503853, "clip_ratio/low_mean": 0.033337948843836784, "clip_ratio/low_min": 0.033337948843836784, "clip_ratio/high_mean": 0.09912267606705427, "clip_ratio/high_max": 0.09912267606705427, "clip_ratio/region_mean": 0.13246062491089106, "reward_total_mean": 0.7830727100372314, "reward_meter_mean": 0.9869276285171509, "reward_meter_std": 0.016702743247151375, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9433026313781738, "reward_repeat_soft_std": 0.023325536400079727, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7830727100372314, "reward_total_composite_std": 0.017859673127532005} {"timestamp_utc": "2026-04-13T01:21:48Z", "mode": "train", "global_step": 1231, "epoch": 0.12365645404319438, "loss": 0.0177, "grad_norm": 7.61756706237793, "learning_rate": 6.2727272727272734e-06, "num_tokens": 2209385.0, "completions/mean_length": 124.125, "completions/min_length": 117.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 124.125, "completions/min_terminated_length": 117.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.9051387906074524, "rewards/meter/std": 0.23109756410121918, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.966147780418396, "rewards/repeat_soft/std": 0.01984497904777527, "rewards/judge_quality/mean": 0.5699999928474426, "rewards/judge_quality/std": 0.16035676002502441, "rewards/total_composite/mean": 0.7999272346496582, "rewards/total_composite/std": 0.09400378912687302, "reward": 0.7999272346496582, "reward_std": 0.09400378912687302, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1444815844297409, "sampling/sampling_logp_difference/max": 1.6972379684448242, "sampling/importance_sampling_ratio/min": 0.183188796043396, "sampling/importance_sampling_ratio/mean": 1.0156595706939697, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.305025964975357, "clip_ratio/low_mean": 0.06964312586933374, "clip_ratio/low_min": 0.06964312586933374, "clip_ratio/high_mean": 0.05237648542970419, "clip_ratio/high_max": 0.05237648542970419, "clip_ratio/region_mean": 0.12201961129903793, "reward_total_mean": 0.7999272346496582, "reward_meter_mean": 0.9051387906074524, "reward_meter_std": 0.23109756410121918, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.966147780418396, "reward_repeat_soft_std": 0.01984497904777527, "reward_judge_quality_mean": 0.5699999928474426, "reward_judge_quality_std": 0.16035676002502441, "reward_total_composite_mean": 0.7999272346496582, "reward_total_composite_std": 0.09400378912687302} {"timestamp_utc": "2026-04-13T01:22:00Z", "mode": "train", "global_step": 1232, "epoch": 0.12375690607734807, "loss": -0.1012, "grad_norm": 3.7822866439819336, "learning_rate": 6.26969696969697e-06, "num_tokens": 2211021.0, "completions/mean_length": 100.5, "completions/min_length": 40.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 41.71428680419922, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.7435267567634583, "rewards/meter/std": 0.4456069767475128, "rewards/count_adherence/mean": 0.5833333730697632, "rewards/count_adherence/std": 0.2357022762298584, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9743236303329468, "rewards/repeat_soft/std": 0.03526224568486214, "rewards/judge_quality/mean": 0.4112499952316284, "rewards/judge_quality/std": 0.1797965168952942, "rewards/total_composite/mean": 0.6285194158554077, "rewards/total_composite/std": 0.2991681694984436, "reward": 0.6285194158554077, "reward_std": 0.2991681396961212, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1548256129026413, "sampling/sampling_logp_difference/max": 1.9578018188476562, "sampling/importance_sampling_ratio/min": 0.14116838574409485, "sampling/importance_sampling_ratio/mean": 1.0330216884613037, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.983751654624939, "clip_ratio/low_mean": 0.017045455053448677, "clip_ratio/low_min": 0.017045455053448677, "clip_ratio/high_mean": 0.10603510215878487, "clip_ratio/high_max": 0.10603510215878487, "clip_ratio/region_mean": 0.12308055721223354, "reward_total_mean": 0.6285194158554077, "reward_meter_mean": 0.7435267567634583, "reward_meter_std": 0.4456069767475128, "reward_count_adherence_mean": 0.5833333730697632, "reward_count_adherence_std": 0.2357022762298584, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9743236303329468, "reward_repeat_soft_std": 0.03526224568486214, "reward_judge_quality_mean": 0.4112499952316284, "reward_judge_quality_std": 0.1797965168952942, "reward_total_composite_mean": 0.6285194158554077, "reward_total_composite_std": 0.2991681694984436} {"timestamp_utc": "2026-04-13T01:22:06Z", "mode": "train", "global_step": 1233, "epoch": 0.12385735811150175, "loss": 0.0044, "grad_norm": 15.230827331542969, "learning_rate": 6.266666666666668e-06, "num_tokens": 2212614.0, "completions/mean_length": 40.125, "completions/min_length": 31.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.125, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.36828863620758057, "rewards/meter/std": 0.3912443220615387, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9758337140083313, "rewards/repeat_soft/std": 0.027695709839463234, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.5768132209777832, "rewards/total_composite/std": 0.17936357855796814, "reward": 0.5768132209777832, "reward_std": 0.17936356365680695, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15814808011054993, "sampling/sampling_logp_difference/max": 1.813857078552246, "sampling/importance_sampling_ratio/min": 0.16302412748336792, "sampling/importance_sampling_ratio/mean": 1.0169355869293213, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0320938527584076, "clip_ratio/low_mean": 0.0620822561904788, "clip_ratio/low_min": 0.0620822561904788, "clip_ratio/high_mean": 0.094960100017488, "clip_ratio/high_max": 0.094960100017488, "clip_ratio/region_mean": 0.1570423562079668, "reward_total_mean": 0.5768132209777832, "reward_meter_mean": 0.36828863620758057, "reward_meter_std": 0.3912443220615387, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9758337140083313, "reward_repeat_soft_std": 0.027695709839463234, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.5768132209777832, "reward_total_composite_std": 0.17936357855796814} {"timestamp_utc": "2026-04-13T01:22:12Z", "mode": "train", "global_step": 1234, "epoch": 0.12395781014565545, "loss": 0.0334, "grad_norm": 16.813953399658203, "learning_rate": 6.263636363636364e-06, "num_tokens": 2214198.0, "completions/mean_length": 31.0, "completions/min_length": 25.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.0, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.5894812941551208, "rewards/meter/std": 0.37620049715042114, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9833714962005615, "rewards/repeat_soft/std": 0.01135347317904234, "rewards/judge_quality/mean": 0.7112500667572021, "rewards/judge_quality/std": 0.2928401827812195, "rewards/total_composite/mean": 0.6319434642791748, "rewards/total_composite/std": 0.32020413875579834, "reward": 0.6319434642791748, "reward_std": 0.32020410895347595, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17601071298122406, "sampling/sampling_logp_difference/max": 2.012850046157837, "sampling/importance_sampling_ratio/min": 0.13360734283924103, "sampling/importance_sampling_ratio/mean": 0.9807050824165344, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0457190424203873, "clip_ratio/low_mean": 0.05220986856147647, "clip_ratio/low_min": 0.05220986856147647, "clip_ratio/high_mean": 0.09286405704915524, "clip_ratio/high_max": 0.09286405704915524, "clip_ratio/region_mean": 0.1450739256106317, "reward_total_mean": 0.6319434642791748, "reward_meter_mean": 0.5894812941551208, "reward_meter_std": 0.37620049715042114, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9833714962005615, "reward_repeat_soft_std": 0.01135347317904234, "reward_judge_quality_mean": 0.7112500667572021, "reward_judge_quality_std": 0.2928401827812195, "reward_total_composite_mean": 0.6319434642791748, "reward_total_composite_std": 0.32020413875579834} {"timestamp_utc": "2026-04-13T01:22:19Z", "mode": "train", "global_step": 1235, "epoch": 0.12405826217980914, "loss": 0.0053, "grad_norm": 13.257739067077637, "learning_rate": 6.260606060606062e-06, "num_tokens": 2215903.0, "completions/mean_length": 45.125, "completions/min_length": 40.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.125, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.8004817366600037, "rewards/meter/std": 0.2707938849925995, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9959697723388672, "rewards/repeat_soft/std": 0.008175122551620007, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.19334925711154938, "rewards/total_composite/mean": 0.7691887617111206, "rewards/total_composite/std": 0.12549325823783875, "reward": 0.7691887617111206, "reward_std": 0.12549324333667755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1481127142906189, "sampling/sampling_logp_difference/max": 1.9108319282531738, "sampling/importance_sampling_ratio/min": 0.1479572355747223, "sampling/importance_sampling_ratio/mean": 1.0259804725646973, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.879877582192421, "clip_ratio/low_mean": 0.04295419156551361, "clip_ratio/low_min": 0.04295419156551361, "clip_ratio/high_mean": 0.09614671114832163, "clip_ratio/high_max": 0.09614671114832163, "clip_ratio/region_mean": 0.13910090271383524, "reward_total_mean": 0.7691887617111206, "reward_meter_mean": 0.8004817366600037, "reward_meter_std": 0.2707938849925995, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9959697723388672, "reward_repeat_soft_std": 0.008175122551620007, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.19334925711154938, "reward_total_composite_mean": 0.7691887617111206, "reward_total_composite_std": 0.12549325823783875} {"timestamp_utc": "2026-04-13T01:22:27Z", "mode": "train", "global_step": 1236, "epoch": 0.12415871421396284, "loss": 0.0878, "grad_norm": 21.926219940185547, "learning_rate": 6.257575757575758e-06, "num_tokens": 2217571.0, "completions/mean_length": 29.5, "completions/min_length": 26.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.5, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.5753853917121887, "rewards/meter/std": 0.4519827961921692, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9790406227111816, "rewards/repeat_soft/std": 0.017768749967217445, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.5775774717330933, "rewards/total_composite/std": 0.19802124798297882, "reward": 0.5775774717330933, "reward_std": 0.19802123308181763, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23454074561595917, "sampling/sampling_logp_difference/max": 3.5937416553497314, "sampling/importance_sampling_ratio/min": 0.027495259419083595, "sampling/importance_sampling_ratio/mean": 0.99915611743927, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9853659644722939, "clip_ratio/low_mean": 0.052703763358294964, "clip_ratio/low_min": 0.052703763358294964, "clip_ratio/high_mean": 0.11546497885137796, "clip_ratio/high_max": 0.11546497885137796, "clip_ratio/region_mean": 0.16816874220967293, "reward_total_mean": 0.5775774717330933, "reward_meter_mean": 0.5753853917121887, "reward_meter_std": 0.4519827961921692, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9790406227111816, "reward_repeat_soft_std": 0.017768749967217445, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.5775774717330933, "reward_total_composite_std": 0.19802124798297882} {"timestamp_utc": "2026-04-13T01:22:33Z", "mode": "train", "global_step": 1237, "epoch": 0.12425916624811652, "loss": 0.0091, "grad_norm": 12.805747985839844, "learning_rate": 6.254545454545455e-06, "num_tokens": 2219431.0, "completions/mean_length": 46.5, "completions/min_length": 42.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.5, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.8835947513580322, "rewards/meter/std": 0.2914412021636963, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9915289878845215, "rewards/repeat_soft/std": 0.005520847626030445, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7238955497741699, "rewards/total_composite/std": 0.13132251799106598, "reward": 0.7238955497741699, "reward_std": 0.13132251799106598, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15279905498027802, "sampling/sampling_logp_difference/max": 1.0924758911132812, "sampling/importance_sampling_ratio/min": 0.3353850841522217, "sampling/importance_sampling_ratio/mean": 1.0254526138305664, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3159327134490013, "clip_ratio/low_mean": 0.014204545877873898, "clip_ratio/low_min": 0.014204545877873898, "clip_ratio/high_mean": 0.1272086501121521, "clip_ratio/high_max": 0.1272086501121521, "clip_ratio/region_mean": 0.141413195990026, "reward_total_mean": 0.7238955497741699, "reward_meter_mean": 0.8835947513580322, "reward_meter_std": 0.2914412021636963, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9915289878845215, "reward_repeat_soft_std": 0.005520847626030445, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7238955497741699, "reward_total_composite_std": 0.13132251799106598} {"timestamp_utc": "2026-04-13T01:22:40Z", "mode": "train", "global_step": 1238, "epoch": 0.12435961828227021, "loss": 0.0383, "grad_norm": 9.247344970703125, "learning_rate": 6.251515151515152e-06, "num_tokens": 2221190.0, "completions/mean_length": 49.875, "completions/min_length": 47.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.875, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.9715063571929932, "rewards/meter/std": 0.06267703324556351, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9180065393447876, "rewards/repeat_soft/std": 0.051978886127471924, "rewards/judge_quality/mean": 0.5012500286102295, "rewards/judge_quality/std": 0.16974246501922607, "rewards/total_composite/mean": 0.8293535709381104, "rewards/total_composite/std": 0.062814861536026, "reward": 0.8293535709381104, "reward_std": 0.0628148689866066, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13775643706321716, "sampling/sampling_logp_difference/max": 1.6688385009765625, "sampling/importance_sampling_ratio/min": 0.18846583366394043, "sampling/importance_sampling_ratio/mean": 1.0247677564620972, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2054442316293716, "clip_ratio/low_mean": 0.0654032789170742, "clip_ratio/low_min": 0.0654032789170742, "clip_ratio/high_mean": 0.03734276816248894, "clip_ratio/high_max": 0.03734276816248894, "clip_ratio/region_mean": 0.10274604707956314, "reward_total_mean": 0.8293535709381104, "reward_meter_mean": 0.9715063571929932, "reward_meter_std": 0.06267703324556351, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9180065393447876, "reward_repeat_soft_std": 0.051978886127471924, "reward_judge_quality_mean": 0.5012500286102295, "reward_judge_quality_std": 0.16974246501922607, "reward_total_composite_mean": 0.8293535709381104, "reward_total_composite_std": 0.062814861536026} {"timestamp_utc": "2026-04-13T01:22:48Z", "mode": "train", "global_step": 1239, "epoch": 0.12446007031642391, "loss": 0.0342, "grad_norm": 6.959392547607422, "learning_rate": 6.248484848484849e-06, "num_tokens": 2223819.0, "completions/mean_length": 129.625, "completions/min_length": 117.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 129.625, "completions/min_terminated_length": 117.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.9938325881958008, "rewards/meter/std": 0.0028607184067368507, "rewards/count_adherence/mean": 0.8541666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7716197967529297, "rewards/repeat_soft/std": 0.07262208312749863, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.7556366324424744, "rewards/total_composite/std": 0.03657589852809906, "reward": 0.7556366324424744, "reward_std": 0.03657589480280876, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10165148973464966, "sampling/sampling_logp_difference/max": 2.03488826751709, "sampling/importance_sampling_ratio/min": 0.13069508969783783, "sampling/importance_sampling_ratio/mean": 1.0151028633117676, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8029562011361122, "clip_ratio/low_mean": 0.016151725314557552, "clip_ratio/low_min": 0.016151725314557552, "clip_ratio/high_mean": 0.050527545157819986, "clip_ratio/high_max": 0.050527545157819986, "clip_ratio/region_mean": 0.06667927047237754, "reward_total_mean": 0.7556366324424744, "reward_meter_mean": 0.9938325881958008, "reward_meter_std": 0.0028607184067368507, "reward_count_adherence_mean": 0.8541666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7716197967529297, "reward_repeat_soft_std": 0.07262208312749863, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.7556366324424744, "reward_total_composite_std": 0.03657589852809906} {"timestamp_utc": "2026-04-13T01:22:55Z", "mode": "train", "global_step": 1240, "epoch": 0.1245605223505776, "loss": 0.0301, "grad_norm": 14.186257362365723, "learning_rate": 6.245454545454545e-06, "num_tokens": 2225470.0, "completions/mean_length": 46.375, "completions/min_length": 44.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.375, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.8579627275466919, "rewards/meter/std": 0.31577447056770325, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9646937847137451, "rewards/repeat_soft/std": 0.02190197817981243, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404788017273, "rewards/total_composite/mean": 0.7994276285171509, "rewards/total_composite/std": 0.10522134602069855, "reward": 0.7994276285171509, "reward_std": 0.10522134602069855, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13203401863574982, "sampling/sampling_logp_difference/max": 1.1739773750305176, "sampling/importance_sampling_ratio/min": 0.30913496017456055, "sampling/importance_sampling_ratio/mean": 1.0182894468307495, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0089142546057701, "clip_ratio/low_mean": 0.03787878900766373, "clip_ratio/low_min": 0.03787878900766373, "clip_ratio/high_mean": 0.10187983699142933, "clip_ratio/high_max": 0.10187983699142933, "clip_ratio/region_mean": 0.13975862599909306, "reward_total_mean": 0.7994276285171509, "reward_meter_mean": 0.8579627275466919, "reward_meter_std": 0.31577447056770325, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9646937847137451, "reward_repeat_soft_std": 0.02190197817981243, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404788017273, "reward_total_composite_mean": 0.7994276285171509, "reward_total_composite_std": 0.10522134602069855} {"timestamp_utc": "2026-04-13T01:23:01Z", "mode": "train", "global_step": 1241, "epoch": 0.1246609743847313, "loss": 0.0584, "grad_norm": 12.371113777160645, "learning_rate": 6.2424242424242434e-06, "num_tokens": 2227070.0, "completions/mean_length": 48.0, "completions/min_length": 39.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.0, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.974225640296936, "rewards/meter/std": 0.038196537643671036, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9656323194503784, "rewards/repeat_soft/std": 0.039763953536748886, "rewards/judge_quality/mean": 0.6399999856948853, "rewards/judge_quality/std": 0.1791248768568039, "rewards/total_composite/mean": 0.876964807510376, "rewards/total_composite/std": 0.053375229239463806, "reward": 0.876964807510376, "reward_std": 0.0533752366900444, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12185642123222351, "sampling/sampling_logp_difference/max": 1.3918962478637695, "sampling/importance_sampling_ratio/min": 0.2486034482717514, "sampling/importance_sampling_ratio/mean": 1.0169825553894043, "sampling/importance_sampling_ratio/max": 1.6840882301330566, "entropy": 1.0742911025881767, "clip_ratio/low_mean": 0.045797599479556084, "clip_ratio/low_min": 0.045797599479556084, "clip_ratio/high_mean": 0.06825413461774588, "clip_ratio/high_max": 0.06825413461774588, "clip_ratio/region_mean": 0.11405173409730196, "reward_total_mean": 0.876964807510376, "reward_meter_mean": 0.974225640296936, "reward_meter_std": 0.038196537643671036, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9656323194503784, "reward_repeat_soft_std": 0.039763953536748886, "reward_judge_quality_mean": 0.6399999856948853, "reward_judge_quality_std": 0.1791248768568039, "reward_total_composite_mean": 0.876964807510376, "reward_total_composite_std": 0.053375229239463806} {"timestamp_utc": "2026-04-13T01:23:13Z", "mode": "train", "global_step": 1242, "epoch": 0.12476142641888498, "loss": -0.1261, "grad_norm": 2.033144950866699, "learning_rate": 6.23939393939394e-06, "num_tokens": 2228751.0, "completions/mean_length": 103.125, "completions/min_length": 42.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 44.71428680419922, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.8712558150291443, "rewards/meter/std": 0.2566814422607422, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9910175800323486, "rewards/repeat_soft/std": 0.007803088985383511, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.24063904583454132, "rewards/total_composite/mean": 0.7994168996810913, "rewards/total_composite/std": 0.17481791973114014, "reward": 0.7994168996810913, "reward_std": 0.17481790482997894, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16336700320243835, "sampling/sampling_logp_difference/max": 1.634624719619751, "sampling/importance_sampling_ratio/min": 0.19502553343772888, "sampling/importance_sampling_ratio/mean": 1.0204905271530151, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.909593440592289, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.13603312661871314, "clip_ratio/high_max": 0.13603312661871314, "clip_ratio/region_mean": 0.13603312661871314, "reward_total_mean": 0.7994168996810913, "reward_meter_mean": 0.8712558150291443, "reward_meter_std": 0.2566814422607422, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9910175800323486, "reward_repeat_soft_std": 0.007803088985383511, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.24063904583454132, "reward_total_composite_mean": 0.7994168996810913, "reward_total_composite_std": 0.17481791973114014} {"timestamp_utc": "2026-04-13T01:23:25Z", "mode": "train", "global_step": 1243, "epoch": 0.12486187845303867, "loss": -0.1702, "grad_norm": 2.5176279544830322, "learning_rate": 6.236363636363637e-06, "num_tokens": 2230899.0, "completions/mean_length": 197.5, "completions/min_length": 85.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 92.66667175292969, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.20418857038021088, "rewards/meter/std": 0.12962016463279724, "rewards/count_adherence/mean": 0.7250000238418579, "rewards/count_adherence/std": 0.14880476891994476, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9683592319488525, "rewards/repeat_soft/std": 0.020633792504668236, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.22025959193706512, "rewards/total_composite/mean": 0.38387876749038696, "rewards/total_composite/std": 0.18371742963790894, "reward": 0.38387876749038696, "reward_std": 0.18371741473674774, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1662014275789261, "sampling/sampling_logp_difference/max": 1.4156532287597656, "sampling/importance_sampling_ratio/min": 0.24276697635650635, "sampling/importance_sampling_ratio/mean": 1.0195682048797607, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.089207500219345, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.1223969804123044, "clip_ratio/high_max": 0.1223969804123044, "clip_ratio/region_mean": 0.1223969804123044, "reward_total_mean": 0.38387876749038696, "reward_meter_mean": 0.20418857038021088, "reward_meter_std": 0.12962016463279724, "reward_count_adherence_mean": 0.7250000238418579, "reward_count_adherence_std": 0.14880476891994476, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9683592319488525, "reward_repeat_soft_std": 0.020633792504668236, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.22025959193706512, "reward_total_composite_mean": 0.38387876749038696, "reward_total_composite_std": 0.18371742963790894} {"timestamp_utc": "2026-04-13T01:23:33Z", "mode": "train", "global_step": 1244, "epoch": 0.12496233048719237, "loss": 0.0129, "grad_norm": 11.109601974487305, "learning_rate": 6.2333333333333335e-06, "num_tokens": 2233084.0, "completions/mean_length": 74.125, "completions/min_length": 69.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.125, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9891599416732788, "rewards/meter/std": 0.008630889467895031, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9471204280853271, "rewards/repeat_soft/std": 0.0523139163851738, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7783340215682983, "rewards/total_composite/std": 0.008999155834317207, "reward": 0.7783340215682983, "reward_std": 0.008999151177704334, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14491820335388184, "sampling/sampling_logp_difference/max": 2.039212226867676, "sampling/importance_sampling_ratio/min": 0.13013118505477905, "sampling/importance_sampling_ratio/mean": 1.0199878215789795, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1176826655864716, "clip_ratio/low_mean": 0.030286793131381273, "clip_ratio/low_min": 0.030286793131381273, "clip_ratio/high_mean": 0.08328216150403023, "clip_ratio/high_max": 0.08328216150403023, "clip_ratio/region_mean": 0.1135689546354115, "reward_total_mean": 0.7783340215682983, "reward_meter_mean": 0.9891599416732788, "reward_meter_std": 0.008630889467895031, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9471204280853271, "reward_repeat_soft_std": 0.0523139163851738, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7783340215682983, "reward_total_composite_std": 0.008999155834317207} {"timestamp_utc": "2026-04-13T01:23:40Z", "mode": "train", "global_step": 1245, "epoch": 0.12506278252134606, "loss": 0.0274, "grad_norm": 24.146669387817383, "learning_rate": 6.230303030303031e-06, "num_tokens": 2234406.0, "completions/mean_length": 20.25, "completions/min_length": 17.0, "completions/max_length": 22.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.25, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 22.0, "rewards/meter/mean": 0.9817837476730347, "rewards/meter/std": 0.020530788227915764, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9518852233886719, "rewards/repeat_soft/std": 0.019805528223514557, "rewards/judge_quality/mean": 0.44624999165534973, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8208662271499634, "rewards/total_composite/std": 0.008565776981413364, "reward": 0.8208662271499634, "reward_std": 0.008565771393477917, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1154891699552536, "sampling/sampling_logp_difference/max": 1.2087054252624512, "sampling/importance_sampling_ratio/min": 0.2985835671424866, "sampling/importance_sampling_ratio/mean": 0.9872764348983765, "sampling/importance_sampling_ratio/max": 1.5189663171768188, "entropy": 1.0710762441158295, "clip_ratio/low_mean": 0.038220551796257496, "clip_ratio/low_min": 0.038220551796257496, "clip_ratio/high_mean": 0.11847466696053743, "clip_ratio/high_max": 0.11847466696053743, "clip_ratio/region_mean": 0.15669521875679493, "reward_total_mean": 0.8208662271499634, "reward_meter_mean": 0.9817837476730347, "reward_meter_std": 0.020530788227915764, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9518852233886719, "reward_repeat_soft_std": 0.019805528223514557, "reward_judge_quality_mean": 0.44624999165534973, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8208662271499634, "reward_total_composite_std": 0.008565776981413364} {"timestamp_utc": "2026-04-13T01:23:46Z", "mode": "train", "global_step": 1246, "epoch": 0.12516323455549974, "loss": 0.0228, "grad_norm": 16.860652923583984, "learning_rate": 6.227272727272727e-06, "num_tokens": 2235740.0, "completions/mean_length": 22.75, "completions/min_length": 21.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.75, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.9942210912704468, "rewards/meter/std": 0.0018505305051803589, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9544185400009155, "rewards/repeat_soft/std": 0.021322686225175858, "rewards/judge_quality/mean": 0.5012500286102295, "rewards/judge_quality/std": 0.16974246501922607, "rewards/total_composite/mean": 0.8432164192199707, "rewards/total_composite/std": 0.05152350291609764, "reward": 0.8432164192199707, "reward_std": 0.05152348801493645, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14148138463497162, "sampling/sampling_logp_difference/max": 0.9997696876525879, "sampling/importance_sampling_ratio/min": 0.3679642081260681, "sampling/importance_sampling_ratio/mean": 1.0266872644424438, "sampling/importance_sampling_ratio/max": 1.8658616542816162, "entropy": 1.2892157286405563, "clip_ratio/low_mean": 0.07596779242157936, "clip_ratio/low_min": 0.07596779242157936, "clip_ratio/high_mean": 0.005681818351149559, "clip_ratio/high_max": 0.005681818351149559, "clip_ratio/region_mean": 0.08164961077272892, "reward_total_mean": 0.8432164192199707, "reward_meter_mean": 0.9942210912704468, "reward_meter_std": 0.0018505305051803589, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9544185400009155, "reward_repeat_soft_std": 0.021322686225175858, "reward_judge_quality_mean": 0.5012500286102295, "reward_judge_quality_std": 0.16974246501922607, "reward_total_composite_mean": 0.8432164192199707, "reward_total_composite_std": 0.05152350291609764} {"timestamp_utc": "2026-04-13T01:23:54Z", "mode": "train", "global_step": 1247, "epoch": 0.12526368658965345, "loss": 0.017, "grad_norm": 11.310413360595703, "learning_rate": 6.224242424242425e-06, "num_tokens": 2237719.0, "completions/mean_length": 72.375, "completions/min_length": 67.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.375, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.5459434986114502, "rewards/meter/std": 0.3364745080471039, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9762474298477173, "rewards/repeat_soft/std": 0.012377824634313583, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.575424313545227, "rewards/total_composite/std": 0.143826425075531, "reward": 0.575424313545227, "reward_std": 0.1438264399766922, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15566392242908478, "sampling/sampling_logp_difference/max": 1.358006477355957, "sampling/importance_sampling_ratio/min": 0.25717294216156006, "sampling/importance_sampling_ratio/mean": 1.0218003988265991, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1627653390169144, "clip_ratio/low_mean": 0.09444881230592728, "clip_ratio/low_min": 0.09444881230592728, "clip_ratio/high_mean": 0.04954605083912611, "clip_ratio/high_max": 0.04954605083912611, "clip_ratio/region_mean": 0.1439948631450534, "reward_total_mean": 0.575424313545227, "reward_meter_mean": 0.5459434986114502, "reward_meter_std": 0.3364745080471039, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9762474298477173, "reward_repeat_soft_std": 0.012377824634313583, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.575424313545227, "reward_total_composite_std": 0.143826425075531} {"timestamp_utc": "2026-04-13T01:24:01Z", "mode": "train", "global_step": 1248, "epoch": 0.12536413862380713, "loss": 0.0388, "grad_norm": 17.64260482788086, "learning_rate": 6.221212121212121e-06, "num_tokens": 2239461.0, "completions/mean_length": 47.75, "completions/min_length": 41.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.75, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.8323096036911011, "rewards/meter/std": 0.261613130569458, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.11785111576318741, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.966842770576477, "rewards/repeat_soft/std": 0.04672551155090332, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.7158485651016235, "rewards/total_composite/std": 0.1146441325545311, "reward": 0.7158485651016235, "reward_std": 0.1146441251039505, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14646707475185394, "sampling/sampling_logp_difference/max": 1.6216959953308105, "sampling/importance_sampling_ratio/min": 0.19756335020065308, "sampling/importance_sampling_ratio/mean": 1.0305567979812622, "sampling/importance_sampling_ratio/max": 1.9893903732299805, "entropy": 1.1693199425935745, "clip_ratio/low_mean": 0.024298839271068573, "clip_ratio/low_min": 0.024298839271068573, "clip_ratio/high_mean": 0.10832445975393057, "clip_ratio/high_max": 0.10832445975393057, "clip_ratio/region_mean": 0.13262329902499914, "reward_total_mean": 0.7158485651016235, "reward_meter_mean": 0.8323096036911011, "reward_meter_std": 0.261613130569458, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.11785111576318741, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.966842770576477, "reward_repeat_soft_std": 0.04672551155090332, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.7158485651016235, "reward_total_composite_std": 0.1146441325545311} {"timestamp_utc": "2026-04-13T01:24:07Z", "mode": "train", "global_step": 1249, "epoch": 0.1254645906579608, "loss": 0.0618, "grad_norm": 16.605375289916992, "learning_rate": 6.218181818181819e-06, "num_tokens": 2241134.0, "completions/mean_length": 44.125, "completions/min_length": 33.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.125, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.5755573511123657, "rewards/meter/std": 0.289407342672348, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9678260087966919, "rewards/repeat_soft/std": 0.031050194054841995, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.6362834572792053, "rewards/total_composite/std": 0.12797239422798157, "reward": 0.6362834572792053, "reward_std": 0.12797240912914276, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15429197251796722, "sampling/sampling_logp_difference/max": 1.613205075263977, "sampling/importance_sampling_ratio/min": 0.19924798607826233, "sampling/importance_sampling_ratio/mean": 1.01790452003479, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1952424719929695, "clip_ratio/low_mean": 0.06643335614353418, "clip_ratio/low_min": 0.06643335614353418, "clip_ratio/high_mean": 0.10874792840331793, "clip_ratio/high_max": 0.10874792840331793, "clip_ratio/region_mean": 0.1751812845468521, "reward_total_mean": 0.6362834572792053, "reward_meter_mean": 0.5755573511123657, "reward_meter_std": 0.289407342672348, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9678260087966919, "reward_repeat_soft_std": 0.031050194054841995, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.6362834572792053, "reward_total_composite_std": 0.12797239422798157} {"timestamp_utc": "2026-04-13T01:24:14Z", "mode": "train", "global_step": 1250, "epoch": 0.12556504269211452, "loss": 0.0114, "grad_norm": 14.588850021362305, "learning_rate": 6.215151515151515e-06, "num_tokens": 2242774.0, "completions/mean_length": 46.0, "completions/min_length": 43.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.0, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9372420310974121, "rewards/meter/std": 0.13551989197731018, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.983672022819519, "rewards/repeat_soft/std": 0.016849718987941742, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.8160011172294617, "rewards/total_composite/std": 0.08359109610319138, "reward": 0.8160011172294617, "reward_std": 0.08359107375144958, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1319725066423416, "sampling/sampling_logp_difference/max": 1.3083007335662842, "sampling/importance_sampling_ratio/min": 0.2702789306640625, "sampling/importance_sampling_ratio/mean": 1.0162668228149414, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8950765654444695, "clip_ratio/low_mean": 0.010869565419852734, "clip_ratio/low_min": 0.010869565419852734, "clip_ratio/high_mean": 0.13269882090389729, "clip_ratio/high_max": 0.13269882090389729, "clip_ratio/region_mean": 0.14356838632375002, "reward_total_mean": 0.8160011172294617, "reward_meter_mean": 0.9372420310974121, "reward_meter_std": 0.13551989197731018, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.983672022819519, "reward_repeat_soft_std": 0.016849718987941742, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.8160011172294617, "reward_total_composite_std": 0.08359109610319138} {"timestamp_utc": "2026-04-13T01:25:00Z", "mode": "eval", "global_step": 1250, "epoch": 0.12556504269211452, "eval_loss": NaN, "eval_runtime": 45.9883, "eval_samples_per_second": 1.74, "eval_steps_per_second": 0.217, "eval_num_tokens": 2242774.0, "eval_completions/mean_length": 69.4625, "eval_completions/min_length": 30.4, "eval_completions/max_length": 145.3, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 57.816666793823245, "eval_completions/min_terminated_length": 30.4, "eval_completions/max_terminated_length": 101.5, "eval_rewards/meter/mean": 0.7925770819187165, "eval_rewards/meter/std": 0.30876830965280533, "eval_rewards/count_adherence/mean": 0.846875, "eval_rewards/count_adherence/std": 0.12768454775214194, "eval_rewards/hard_gate/mean": 0.9625, "eval_rewards/hard_gate/std": 0.0816463440656662, "eval_rewards/repeat_soft/mean": 0.9658715128898621, "eval_rewards/repeat_soft/std": 0.031621781177818775, "eval_rewards/judge_quality/mean": 0.45462500154972074, "eval_rewards/judge_quality/std": 0.1277757921256125, "eval_rewards/total_composite/mean": 0.7032939851284027, "eval_rewards/total_composite/std": 0.18225929588079454, "eval_reward": 0.7032939851284027, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.08944797068834305, "eval_sampling/sampling_logp_difference/max": 1.0509572982788087, "eval_sampling/importance_sampling_ratio/min": 0.3610517933964729, "eval_sampling/importance_sampling_ratio/mean": 1.0273959398269654, "eval_sampling/importance_sampling_ratio/max": 1.4386500477790833, "eval_entropy": 1.1045615017414092, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7032939851284027, "eval_reward_meter_mean": 0.7925770819187165, "eval_reward_meter_std": 0.30876830965280533, "eval_reward_count_adherence_mean": 0.846875, "eval_reward_count_adherence_std": 0.12768454775214194, "eval_reward_hard_gate_mean": 0.9625, "eval_reward_hard_gate_std": 0.0816463440656662, "eval_reward_repeat_soft_mean": 0.9658715128898621, "eval_reward_repeat_soft_std": 0.031621781177818775, "eval_reward_judge_quality_mean": 0.45462500154972074, "eval_reward_judge_quality_std": 0.1277757921256125, "eval_reward_total_composite_mean": 0.7032939851284027, "eval_reward_total_composite_std": 0.18225929588079454} {"timestamp_utc": "2026-04-13T01:25:10Z", "mode": "train", "global_step": 1251, "epoch": 0.1256654947262682, "loss": 0.0039, "grad_norm": 12.084244728088379, "learning_rate": 6.212121212121213e-06, "num_tokens": 2244504.0, "completions/mean_length": 47.25, "completions/min_length": 43.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.25, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.8458865880966187, "rewards/meter/std": 0.20230631530284882, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9912033081054688, "rewards/repeat_soft/std": 0.01029597595334053, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7113943099975586, "rewards/total_composite/std": 0.092808298766613, "reward": 0.7113943099975586, "reward_std": 0.092808298766613, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13906720280647278, "sampling/sampling_logp_difference/max": 1.2618765830993652, "sampling/importance_sampling_ratio/min": 0.2831222116947174, "sampling/importance_sampling_ratio/mean": 1.026245355606079, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0331771224737167, "clip_ratio/low_mean": 0.005319148767739534, "clip_ratio/low_min": 0.005319148767739534, "clip_ratio/high_mean": 0.10916589247062802, "clip_ratio/high_max": 0.10916589247062802, "clip_ratio/region_mean": 0.11448504123836756, "reward_total_mean": 0.7113943099975586, "reward_meter_mean": 0.8458865880966187, "reward_meter_std": 0.20230631530284882, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9912033081054688, "reward_repeat_soft_std": 0.01029597595334053, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7113943099975586, "reward_total_composite_std": 0.092808298766613} {"timestamp_utc": "2026-04-13T01:25:17Z", "mode": "train", "global_step": 1252, "epoch": 0.1257659467604219, "loss": 0.023, "grad_norm": 14.665938377380371, "learning_rate": 6.209090909090909e-06, "num_tokens": 2246085.0, "completions/mean_length": 35.625, "completions/min_length": 31.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.625, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9594019055366516, "rewards/meter/std": 0.043721020221710205, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.966945230960846, "rewards/repeat_soft/std": 0.027801793068647385, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.10392305999994278, "rewards/total_composite/mean": 0.8179253339767456, "rewards/total_composite/std": 0.040430162101984024, "reward": 0.8179253339767456, "reward_std": 0.04043015465140343, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15154485404491425, "sampling/sampling_logp_difference/max": 1.579483985900879, "sampling/importance_sampling_ratio/min": 0.20608140528202057, "sampling/importance_sampling_ratio/mean": 1.0543434619903564, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2580480873584747, "clip_ratio/low_mean": 0.09930113982409239, "clip_ratio/low_min": 0.09930113982409239, "clip_ratio/high_mean": 0.06054505333304405, "clip_ratio/high_max": 0.06054505333304405, "clip_ratio/region_mean": 0.15984619315713644, "reward_total_mean": 0.8179253339767456, "reward_meter_mean": 0.9594019055366516, "reward_meter_std": 0.043721020221710205, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.966945230960846, "reward_repeat_soft_std": 0.027801793068647385, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.10392305999994278, "reward_total_composite_mean": 0.8179253339767456, "reward_total_composite_std": 0.040430162101984024} {"timestamp_utc": "2026-04-13T01:25:24Z", "mode": "train", "global_step": 1253, "epoch": 0.1258663987945756, "loss": 0.069, "grad_norm": 12.002691268920898, "learning_rate": 6.206060606060606e-06, "num_tokens": 2247860.0, "completions/mean_length": 46.875, "completions/min_length": 39.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.875, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.5610209703445435, "rewards/meter/std": 0.3212539255619049, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9777528047561646, "rewards/repeat_soft/std": 0.014579824171960354, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.5823597311973572, "rewards/total_composite/std": 0.14711439609527588, "reward": 0.5823597311973572, "reward_std": 0.1471143662929535, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14006270468235016, "sampling/sampling_logp_difference/max": 1.8545827865600586, "sampling/importance_sampling_ratio/min": 0.15651823580265045, "sampling/importance_sampling_ratio/mean": 0.9969016909599304, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9091097712516785, "clip_ratio/low_mean": 0.0636066934093833, "clip_ratio/low_min": 0.0636066934093833, "clip_ratio/high_mean": 0.09936153329908848, "clip_ratio/high_max": 0.09936153329908848, "clip_ratio/region_mean": 0.16296822670847178, "reward_total_mean": 0.5823597311973572, "reward_meter_mean": 0.5610209703445435, "reward_meter_std": 0.3212539255619049, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9777528047561646, "reward_repeat_soft_std": 0.014579824171960354, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.5823597311973572, "reward_total_composite_std": 0.14711439609527588} {"timestamp_utc": "2026-04-13T01:25:31Z", "mode": "train", "global_step": 1254, "epoch": 0.12596685082872927, "loss": 0.0253, "grad_norm": 10.54057788848877, "learning_rate": 6.203030303030304e-06, "num_tokens": 2249455.0, "completions/mean_length": 42.375, "completions/min_length": 40.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.375, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.7829062938690186, "rewards/meter/std": 0.37797075510025024, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9409051537513733, "rewards/repeat_soft/std": 0.049884337931871414, "rewards/judge_quality/mean": 0.6575000286102295, "rewards/judge_quality/std": 0.2133910059928894, "rewards/total_composite/mean": 0.793648362159729, "rewards/total_composite/std": 0.17504429817199707, "reward": 0.793648362159729, "reward_std": 0.17504431307315826, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1283097118139267, "sampling/sampling_logp_difference/max": 1.2818193435668945, "sampling/importance_sampling_ratio/min": 0.2775318920612335, "sampling/importance_sampling_ratio/mean": 1.030547857284546, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8925838395953178, "clip_ratio/low_mean": 0.020632444880902767, "clip_ratio/low_min": 0.020632444880902767, "clip_ratio/high_mean": 0.08821573480963707, "clip_ratio/high_max": 0.08821573480963707, "clip_ratio/region_mean": 0.10884817969053984, "reward_total_mean": 0.793648362159729, "reward_meter_mean": 0.7829062938690186, "reward_meter_std": 0.37797075510025024, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9409051537513733, "reward_repeat_soft_std": 0.049884337931871414, "reward_judge_quality_mean": 0.6575000286102295, "reward_judge_quality_std": 0.2133910059928894, "reward_total_composite_mean": 0.793648362159729, "reward_total_composite_std": 0.17504429817199707} {"timestamp_utc": "2026-04-13T01:25:37Z", "mode": "train", "global_step": 1255, "epoch": 0.12606730286288298, "loss": 0.0086, "grad_norm": 22.789981842041016, "learning_rate": 6.200000000000001e-06, "num_tokens": 2251089.0, "completions/mean_length": 32.25, "completions/min_length": 29.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.25, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.7865064740180969, "rewards/meter/std": 0.18466489017009735, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.11785111576318741, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9904694557189941, "rewards/repeat_soft/std": 0.011617488227784634, "rewards/judge_quality/mean": 0.41749998927116394, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.6844748258590698, "rewards/total_composite/std": 0.08418063819408417, "reward": 0.6844748258590698, "reward_std": 0.08418063819408417, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15210606157779694, "sampling/sampling_logp_difference/max": 1.653404712677002, "sampling/importance_sampling_ratio/min": 0.19139714539051056, "sampling/importance_sampling_ratio/mean": 1.0096585750579834, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9902128428220749, "clip_ratio/low_mean": 0.040526800788939, "clip_ratio/low_min": 0.040526800788939, "clip_ratio/high_mean": 0.07654268108308315, "clip_ratio/high_max": 0.07654268108308315, "clip_ratio/region_mean": 0.11706948187202215, "reward_total_mean": 0.6844748258590698, "reward_meter_mean": 0.7865064740180969, "reward_meter_std": 0.18466489017009735, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.11785111576318741, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9904694557189941, "reward_repeat_soft_std": 0.011617488227784634, "reward_judge_quality_mean": 0.41749998927116394, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.6844748258590698, "reward_total_composite_std": 0.08418063819408417} {"timestamp_utc": "2026-04-13T01:25:44Z", "mode": "train", "global_step": 1256, "epoch": 0.12616775489703666, "loss": 0.017, "grad_norm": 15.171074867248535, "learning_rate": 6.196969696969698e-06, "num_tokens": 2252688.0, "completions/mean_length": 42.875, "completions/min_length": 39.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.875, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.6727914810180664, "rewards/meter/std": 0.3363889455795288, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9564321041107178, "rewards/repeat_soft/std": 0.06940910965204239, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.6713993549346924, "rewards/total_composite/std": 0.15507182478904724, "reward": 0.6713993549346924, "reward_std": 0.15507182478904724, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17049510776996613, "sampling/sampling_logp_difference/max": 1.444458246231079, "sampling/importance_sampling_ratio/min": 0.23587384819984436, "sampling/importance_sampling_ratio/mean": 1.0272667407989502, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3278893157839775, "clip_ratio/low_mean": 0.052145497873425484, "clip_ratio/low_min": 0.052145497873425484, "clip_ratio/high_mean": 0.09983839560300112, "clip_ratio/high_max": 0.09983839560300112, "clip_ratio/region_mean": 0.1519838934764266, "reward_total_mean": 0.6713993549346924, "reward_meter_mean": 0.6727914810180664, "reward_meter_std": 0.3363889455795288, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9564321041107178, "reward_repeat_soft_std": 0.06940910965204239, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.6713993549346924, "reward_total_composite_std": 0.15507182478904724} {"timestamp_utc": "2026-04-13T01:25:51Z", "mode": "train", "global_step": 1257, "epoch": 0.12626820693119037, "loss": -0.0077, "grad_norm": 21.88886070251465, "learning_rate": 6.1939393939393944e-06, "num_tokens": 2254070.0, "completions/mean_length": 22.75, "completions/min_length": 19.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.75, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.880578875541687, "rewards/meter/std": 0.3104945421218872, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9521893262863159, "rewards/repeat_soft/std": 0.016934430226683617, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7742294669151306, "rewards/total_composite/std": 0.1389118731021881, "reward": 0.7742294669151306, "reward_std": 0.1389118731021881, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16836269199848175, "sampling/sampling_logp_difference/max": 1.377549171447754, "sampling/importance_sampling_ratio/min": 0.25219589471817017, "sampling/importance_sampling_ratio/mean": 1.0213507413864136, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0687570720911026, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.14758061151951551, "clip_ratio/high_max": 0.14758061151951551, "clip_ratio/region_mean": 0.14758061151951551, "reward_total_mean": 0.7742294669151306, "reward_meter_mean": 0.880578875541687, "reward_meter_std": 0.3104945421218872, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9521893262863159, "reward_repeat_soft_std": 0.016934430226683617, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7742294669151306, "reward_total_composite_std": 0.1389118731021881} {"timestamp_utc": "2026-04-13T01:26:03Z", "mode": "train", "global_step": 1258, "epoch": 0.12636865896534405, "loss": -0.0334, "grad_norm": 1.0089293718338013, "learning_rate": 6.190909090909092e-06, "num_tokens": 2255326.0, "completions/mean_length": 202.0, "completions/min_length": 14.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 16.0, "completions/min_terminated_length": 14.0, "completions/max_terminated_length": 21.0, "rewards/meter/mean": 0.6179625391960144, "rewards/meter/std": 0.45107436180114746, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9601507186889648, "rewards/repeat_soft/std": 0.025442032143473625, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.360394150018692, "rewards/total_composite/mean": 0.49979615211486816, "rewards/total_composite/std": 0.423373281955719, "reward": 0.49979615211486816, "reward_std": 0.4233732521533966, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09657314419746399, "sampling/sampling_logp_difference/max": 1.5461783409118652, "sampling/importance_sampling_ratio/min": 0.2130606770515442, "sampling/importance_sampling_ratio/mean": 1.0266180038452148, "sampling/importance_sampling_ratio/max": 1.7967997789382935, "entropy": 0.4723391458392143, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.03511904925107956, "clip_ratio/high_max": 0.03511904925107956, "clip_ratio/region_mean": 0.03511904925107956, "reward_total_mean": 0.49979615211486816, "reward_meter_mean": 0.6179625391960144, "reward_meter_std": 0.45107436180114746, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9601507186889648, "reward_repeat_soft_std": 0.025442032143473625, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.360394150018692, "reward_total_composite_mean": 0.49979615211486816, "reward_total_composite_std": 0.423373281955719} {"timestamp_utc": "2026-04-13T01:26:10Z", "mode": "train", "global_step": 1259, "epoch": 0.12646911099949773, "loss": 0.0527, "grad_norm": 12.050110816955566, "learning_rate": 6.187878787878788e-06, "num_tokens": 2257304.0, "completions/mean_length": 67.25, "completions/min_length": 58.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.25, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.974170446395874, "rewards/meter/std": 0.04933997988700867, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9311051368713379, "rewards/repeat_soft/std": 0.09189947694540024, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7572371959686279, "rewards/total_composite/std": 0.04348110780119896, "reward": 0.7572371959686279, "reward_std": 0.04348111152648926, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15954606235027313, "sampling/sampling_logp_difference/max": 1.3665986061096191, "sampling/importance_sampling_ratio/min": 0.2554255723953247, "sampling/importance_sampling_ratio/mean": 1.0370938777923584, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.275550752878189, "clip_ratio/low_mean": 0.04304029420018196, "clip_ratio/low_min": 0.04304029420018196, "clip_ratio/high_mean": 0.09982820227742195, "clip_ratio/high_max": 0.09982820227742195, "clip_ratio/region_mean": 0.1428684964776039, "reward_total_mean": 0.7572371959686279, "reward_meter_mean": 0.974170446395874, "reward_meter_std": 0.04933997988700867, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9311051368713379, "reward_repeat_soft_std": 0.09189947694540024, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7572371959686279, "reward_total_composite_std": 0.04348110780119896} {"timestamp_utc": "2026-04-13T01:26:18Z", "mode": "train", "global_step": 1260, "epoch": 0.12656956303365144, "loss": -0.0103, "grad_norm": 10.75382137298584, "learning_rate": 6.184848484848485e-06, "num_tokens": 2259092.0, "completions/mean_length": 59.5, "completions/min_length": 56.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.5, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9871140718460083, "rewards/meter/std": 0.008727706968784332, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9730416536331177, "rewards/repeat_soft/std": 0.024296434596180916, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7822555303573608, "rewards/total_composite/std": 0.007648279890418053, "reward": 0.7822555303573608, "reward_std": 0.007648269180208445, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16994960606098175, "sampling/sampling_logp_difference/max": 2.12188720703125, "sampling/importance_sampling_ratio/min": 0.11980532109737396, "sampling/importance_sampling_ratio/mean": 1.0320786237716675, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2396584451198578, "clip_ratio/low_mean": 0.0875315461307764, "clip_ratio/low_min": 0.0875315461307764, "clip_ratio/high_mean": 0.06712947227060795, "clip_ratio/high_max": 0.06712947227060795, "clip_ratio/region_mean": 0.15466101840138435, "reward_total_mean": 0.7822555303573608, "reward_meter_mean": 0.9871140718460083, "reward_meter_std": 0.008727706968784332, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9730416536331177, "reward_repeat_soft_std": 0.024296434596180916, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7822555303573608, "reward_total_composite_std": 0.007648279890418053} {"timestamp_utc": "2026-04-13T01:26:25Z", "mode": "train", "global_step": 1261, "epoch": 0.12667001506780512, "loss": -0.0146, "grad_norm": 13.934267044067383, "learning_rate": 6.181818181818182e-06, "num_tokens": 2260865.0, "completions/mean_length": 42.625, "completions/min_length": 37.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.625, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.9533282518386841, "rewards/meter/std": 0.08502477407455444, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9511154294013977, "rewards/repeat_soft/std": 0.04540019482374191, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.8046092987060547, "rewards/total_composite/std": 0.03873005509376526, "reward": 0.8046092987060547, "reward_std": 0.03873005509376526, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13575007021427155, "sampling/sampling_logp_difference/max": 1.3508796691894531, "sampling/importance_sampling_ratio/min": 0.2590123116970062, "sampling/importance_sampling_ratio/mean": 1.0439479351043701, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.064900554716587, "clip_ratio/low_mean": 0.04583845101296902, "clip_ratio/low_min": 0.04583845101296902, "clip_ratio/high_mean": 0.12486273236572742, "clip_ratio/high_max": 0.12486273236572742, "clip_ratio/region_mean": 0.17070118337869644, "reward_total_mean": 0.8046092987060547, "reward_meter_mean": 0.9533282518386841, "reward_meter_std": 0.08502477407455444, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9511154294013977, "reward_repeat_soft_std": 0.04540019482374191, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.8046092987060547, "reward_total_composite_std": 0.03873005509376526} {"timestamp_utc": "2026-04-13T01:26:32Z", "mode": "train", "global_step": 1262, "epoch": 0.12677046710195883, "loss": 0.0827, "grad_norm": 20.93840217590332, "learning_rate": 6.17878787878788e-06, "num_tokens": 2262273.0, "completions/mean_length": 24.0, "completions/min_length": 20.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.0, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.619756817817688, "rewards/meter/std": 0.38065022230148315, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9385402202606201, "rewards/repeat_soft/std": 0.04316587746143341, "rewards/judge_quality/mean": 0.5225000381469727, "rewards/judge_quality/std": 0.264993280172348, "rewards/total_composite/mean": 0.6794945597648621, "rewards/total_composite/std": 0.13208580017089844, "reward": 0.6794945597648621, "reward_std": 0.13208580017089844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13099293410778046, "sampling/sampling_logp_difference/max": 0.9789981842041016, "sampling/importance_sampling_ratio/min": 0.37568730115890503, "sampling/importance_sampling_ratio/mean": 0.9979006052017212, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9559741318225861, "clip_ratio/low_mean": 0.028148148208856583, "clip_ratio/low_min": 0.028148148208856583, "clip_ratio/high_mean": 0.0830341475084424, "clip_ratio/high_max": 0.0830341475084424, "clip_ratio/region_mean": 0.11118229571729898, "reward_total_mean": 0.6794945597648621, "reward_meter_mean": 0.619756817817688, "reward_meter_std": 0.38065022230148315, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9385402202606201, "reward_repeat_soft_std": 0.04316587746143341, "reward_judge_quality_mean": 0.5225000381469727, "reward_judge_quality_std": 0.264993280172348, "reward_total_composite_mean": 0.6794945597648621, "reward_total_composite_std": 0.13208580017089844} {"timestamp_utc": "2026-04-13T01:26:40Z", "mode": "train", "global_step": 1263, "epoch": 0.1268709191361125, "loss": -0.0044, "grad_norm": 9.354107856750488, "learning_rate": 6.175757575757576e-06, "num_tokens": 2264350.0, "completions/mean_length": 76.625, "completions/min_length": 67.0, "completions/max_length": 115.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.625, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 115.0, "rewards/meter/mean": 0.9824119806289673, "rewards/meter/std": 0.012769481167197227, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9609659910202026, "rewards/repeat_soft/std": 0.04418375343084335, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.7926194667816162, "rewards/total_composite/std": 0.0337458960711956, "reward": 0.7926194667816162, "reward_std": 0.033745914697647095, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13858741521835327, "sampling/sampling_logp_difference/max": 1.5735564231872559, "sampling/importance_sampling_ratio/min": 0.20730659365653992, "sampling/importance_sampling_ratio/mean": 1.0378495454788208, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3822873383760452, "clip_ratio/low_mean": 0.11748700402677059, "clip_ratio/low_min": 0.11748700402677059, "clip_ratio/high_mean": 0.02101959567517042, "clip_ratio/high_max": 0.02101959567517042, "clip_ratio/region_mean": 0.138506599701941, "reward_total_mean": 0.7926194667816162, "reward_meter_mean": 0.9824119806289673, "reward_meter_std": 0.012769481167197227, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9609659910202026, "reward_repeat_soft_std": 0.04418375343084335, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.7926194667816162, "reward_total_composite_std": 0.0337458960711956} {"timestamp_utc": "2026-04-13T01:26:47Z", "mode": "train", "global_step": 1264, "epoch": 0.1269713711702662, "loss": -0.1493, "grad_norm": 9.411858558654785, "learning_rate": 6.1727272727272735e-06, "num_tokens": 2266124.0, "completions/mean_length": 62.75, "completions/min_length": 42.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.75, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9889214038848877, "rewards/meter/std": 0.00334029458463192, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.17251639068126678, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9666034579277039, "rewards/repeat_soft/std": 0.024471744894981384, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.8037999868392944, "rewards/total_composite/std": 0.054851289838552475, "reward": 0.8037999868392944, "reward_std": 0.054851289838552475, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14079874753952026, "sampling/sampling_logp_difference/max": 1.3750519752502441, "sampling/importance_sampling_ratio/min": 0.252826452255249, "sampling/importance_sampling_ratio/mean": 1.024409294128418, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1565235033631325, "clip_ratio/low_mean": 0.05873650498688221, "clip_ratio/low_min": 0.05873650498688221, "clip_ratio/high_mean": 0.08861907385289669, "clip_ratio/high_max": 0.08861907385289669, "clip_ratio/region_mean": 0.1473555788397789, "reward_total_mean": 0.8037999868392944, "reward_meter_mean": 0.9889214038848877, "reward_meter_std": 0.00334029458463192, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.17251639068126678, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9666034579277039, "reward_repeat_soft_std": 0.024471744894981384, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.8037999868392944, "reward_total_composite_std": 0.054851289838552475} {"timestamp_utc": "2026-04-13T01:26:59Z", "mode": "train", "global_step": 1265, "epoch": 0.1270718232044199, "loss": -0.1793, "grad_norm": 2.2698426246643066, "learning_rate": 6.16969696969697e-06, "num_tokens": 2268022.0, "completions/mean_length": 130.25, "completions/min_length": 61.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 75.71428680419922, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.8562080264091492, "rewards/meter/std": 0.2855667769908905, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9872320890426636, "rewards/repeat_soft/std": 0.013690100982785225, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.23445607721805573, "rewards/total_composite/mean": 0.6938826441764832, "rewards/total_composite/std": 0.28085067868232727, "reward": 0.6938826441764832, "reward_std": 0.28085067868232727, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14662161469459534, "sampling/sampling_logp_difference/max": 1.4091622829437256, "sampling/importance_sampling_ratio/min": 0.24434790015220642, "sampling/importance_sampling_ratio/mean": 1.0439451932907104, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.19753959774971, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10941911861300468, "clip_ratio/high_max": 0.10941911861300468, "clip_ratio/region_mean": 0.10941911861300468, "reward_total_mean": 0.6938826441764832, "reward_meter_mean": 0.8562080264091492, "reward_meter_std": 0.2855667769908905, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9872320890426636, "reward_repeat_soft_std": 0.013690100982785225, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.23445607721805573, "reward_total_composite_mean": 0.6938826441764832, "reward_total_composite_std": 0.28085067868232727} {"timestamp_utc": "2026-04-13T01:27:11Z", "mode": "train", "global_step": 1266, "epoch": 0.12717227523857358, "loss": -0.0569, "grad_norm": 4.6112494468688965, "learning_rate": 6.166666666666667e-06, "num_tokens": 2269458.0, "completions/mean_length": 99.5, "completions/min_length": 35.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 40.57143020629883, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.3543248176574707, "rewards/meter/std": 0.38895776867866516, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9876164197921753, "rewards/repeat_soft/std": 0.011203274130821228, "rewards/judge_quality/mean": 0.48125001788139343, "rewards/judge_quality/std": 0.2531762719154358, "rewards/total_composite/mean": 0.5432078242301941, "rewards/total_composite/std": 0.21356217563152313, "reward": 0.5432078242301941, "reward_std": 0.21356216073036194, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16082686185836792, "sampling/sampling_logp_difference/max": 1.2406501770019531, "sampling/importance_sampling_ratio/min": 0.2891961336135864, "sampling/importance_sampling_ratio/mean": 1.0455917119979858, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.013918362557888, "clip_ratio/low_mean": 0.0791083937510848, "clip_ratio/low_min": 0.0791083937510848, "clip_ratio/high_mean": 0.056432354263961315, "clip_ratio/high_max": 0.056432354263961315, "clip_ratio/region_mean": 0.13554074801504612, "reward_total_mean": 0.5432078242301941, "reward_meter_mean": 0.3543248176574707, "reward_meter_std": 0.38895776867866516, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9876164197921753, "reward_repeat_soft_std": 0.011203274130821228, "reward_judge_quality_mean": 0.48125001788139343, "reward_judge_quality_std": 0.2531762719154358, "reward_total_composite_mean": 0.5432078242301941, "reward_total_composite_std": 0.21356217563152313} {"timestamp_utc": "2026-04-13T01:27:23Z", "mode": "train", "global_step": 1267, "epoch": 0.12727272727272726, "loss": -0.037, "grad_norm": 1.1639585494995117, "learning_rate": 6.163636363636364e-06, "num_tokens": 2270903.0, "completions/mean_length": 204.625, "completions/min_length": 16.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 20.200000762939453, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.41520798206329346, "rewards/meter/std": 0.4598260223865509, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9642819762229919, "rewards/repeat_soft/std": 0.0304182767868042, "rewards/judge_quality/mean": 0.26374998688697815, "rewards/judge_quality/std": 0.18715444207191467, "rewards/total_composite/mean": 0.3931576907634735, "rewards/total_composite/std": 0.3637184798717499, "reward": 0.3931576907634735, "reward_std": 0.36371850967407227, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18304851651191711, "sampling/sampling_logp_difference/max": 1.421799898147583, "sampling/importance_sampling_ratio/min": 0.24127936363220215, "sampling/importance_sampling_ratio/mean": 1.04150390625, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.046384572982788, "clip_ratio/low_mean": 0.016304347664117813, "clip_ratio/low_min": 0.016304347664117813, "clip_ratio/high_mean": 0.06443452509120107, "clip_ratio/high_max": 0.06443452509120107, "clip_ratio/region_mean": 0.08073887275531888, "reward_total_mean": 0.3931576907634735, "reward_meter_mean": 0.41520798206329346, "reward_meter_std": 0.4598260223865509, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9642819762229919, "reward_repeat_soft_std": 0.0304182767868042, "reward_judge_quality_mean": 0.26374998688697815, "reward_judge_quality_std": 0.18715444207191467, "reward_total_composite_mean": 0.3931576907634735, "reward_total_composite_std": 0.3637184798717499} {"timestamp_utc": "2026-04-13T01:27:31Z", "mode": "train", "global_step": 1268, "epoch": 0.12737317930688097, "loss": 0.0471, "grad_norm": 18.751562118530273, "learning_rate": 6.160606060606062e-06, "num_tokens": 2272463.0, "completions/mean_length": 33.0, "completions/min_length": 27.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.0, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.4767223298549652, "rewards/meter/std": 0.3386731743812561, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9790155291557312, "rewards/repeat_soft/std": 0.02223171852529049, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.6259266138076782, "rewards/total_composite/std": 0.17039088904857635, "reward": 0.6259266138076782, "reward_std": 0.17039090394973755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1639697402715683, "sampling/sampling_logp_difference/max": 1.9116010665893555, "sampling/importance_sampling_ratio/min": 0.14784349501132965, "sampling/importance_sampling_ratio/mean": 0.9964302778244019, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7945691347122192, "clip_ratio/low_mean": 0.08851623255759478, "clip_ratio/low_min": 0.08851623255759478, "clip_ratio/high_mean": 0.06927083432674408, "clip_ratio/high_max": 0.06927083432674408, "clip_ratio/region_mean": 0.15778706688433886, "reward_total_mean": 0.6259266138076782, "reward_meter_mean": 0.4767223298549652, "reward_meter_std": 0.3386731743812561, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9790155291557312, "reward_repeat_soft_std": 0.02223171852529049, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.6259266138076782, "reward_total_composite_std": 0.17039088904857635} {"timestamp_utc": "2026-04-13T01:27:38Z", "mode": "train", "global_step": 1269, "epoch": 0.12747363134103465, "loss": 0.0518, "grad_norm": 16.7665958404541, "learning_rate": 6.157575757575758e-06, "num_tokens": 2273893.0, "completions/mean_length": 32.75, "completions/min_length": 28.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.75, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.499504417181015, "rewards/meter/std": 0.30356329679489136, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.997057318687439, "rewards/repeat_soft/std": 0.005990580189973116, "rewards/judge_quality/mean": 0.7100000381469727, "rewards/judge_quality/std": 0.23862704634666443, "rewards/total_composite/mean": 0.6874827146530151, "rewards/total_composite/std": 0.09499455243349075, "reward": 0.6874827146530151, "reward_std": 0.09499455988407135, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17218826711177826, "sampling/sampling_logp_difference/max": 3.894777297973633, "sampling/importance_sampling_ratio/min": 0.020347904413938522, "sampling/importance_sampling_ratio/mean": 1.0068891048431396, "sampling/importance_sampling_ratio/max": 1.9657433032989502, "entropy": 1.051323376595974, "clip_ratio/low_mean": 0.06547619216144085, "clip_ratio/low_min": 0.06547619216144085, "clip_ratio/high_mean": 0.11928020231425762, "clip_ratio/high_max": 0.11928020231425762, "clip_ratio/region_mean": 0.18475639447569847, "reward_total_mean": 0.6874827146530151, "reward_meter_mean": 0.499504417181015, "reward_meter_std": 0.30356329679489136, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.997057318687439, "reward_repeat_soft_std": 0.005990580189973116, "reward_judge_quality_mean": 0.7100000381469727, "reward_judge_quality_std": 0.23862704634666443, "reward_total_composite_mean": 0.6874827146530151, "reward_total_composite_std": 0.09499455243349075} {"timestamp_utc": "2026-04-13T01:27:45Z", "mode": "train", "global_step": 1270, "epoch": 0.12757408337518836, "loss": 0.0037, "grad_norm": 19.619834899902344, "learning_rate": 6.154545454545455e-06, "num_tokens": 2275704.0, "completions/mean_length": 46.375, "completions/min_length": 44.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.375, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.6135261058807373, "rewards/meter/std": 0.30578726530075073, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9694833755493164, "rewards/repeat_soft/std": 0.039849501103162766, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.6614100933074951, "rewards/total_composite/std": 0.1427186131477356, "reward": 0.6614100933074951, "reward_std": 0.1427186131477356, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12344372272491455, "sampling/sampling_logp_difference/max": 2.512944221496582, "sampling/importance_sampling_ratio/min": 0.08102931827306747, "sampling/importance_sampling_ratio/mean": 1.0393751859664917, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8416187688708305, "clip_ratio/low_mean": 0.05263674072921276, "clip_ratio/low_min": 0.05263674072921276, "clip_ratio/high_mean": 0.044093912933021784, "clip_ratio/high_max": 0.044093912933021784, "clip_ratio/region_mean": 0.09673065366223454, "reward_total_mean": 0.6614100933074951, "reward_meter_mean": 0.6135261058807373, "reward_meter_std": 0.30578726530075073, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9694833755493164, "reward_repeat_soft_std": 0.039849501103162766, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.6614100933074951, "reward_total_composite_std": 0.1427186131477356} {"timestamp_utc": "2026-04-13T01:27:56Z", "mode": "train", "global_step": 1271, "epoch": 0.12767453540934204, "loss": -0.086, "grad_norm": 4.573165416717529, "learning_rate": 6.151515151515152e-06, "num_tokens": 2278234.0, "completions/mean_length": 178.25, "completions/min_length": 111.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 130.57144165039062, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 150.0, "rewards/meter/mean": 0.9571864604949951, "rewards/meter/std": 0.061151694506406784, "rewards/count_adherence/mean": 0.9166666269302368, "rewards/count_adherence/std": 0.0890870913863182, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.913884162902832, "rewards/repeat_soft/std": 0.05960848182439804, "rewards/judge_quality/mean": 0.2849999964237213, "rewards/judge_quality/std": 0.12177261710166931, "rewards/total_composite/mean": 0.4698163866996765, "rewards/total_composite/std": 0.39093929529190063, "reward": 0.4698163866996765, "reward_std": 0.390939325094223, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13323310017585754, "sampling/sampling_logp_difference/max": 3.1760737895965576, "sampling/importance_sampling_ratio/min": 0.0417492501437664, "sampling/importance_sampling_ratio/mean": 1.0155590772628784, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9286548495292664, "clip_ratio/low_mean": 0.02029939368367195, "clip_ratio/low_min": 0.02029939368367195, "clip_ratio/high_mean": 0.0811205143108964, "clip_ratio/high_max": 0.0811205143108964, "clip_ratio/region_mean": 0.10141990799456835, "reward_total_mean": 0.4698163866996765, "reward_meter_mean": 0.9571864604949951, "reward_meter_std": 0.061151694506406784, "reward_count_adherence_mean": 0.9166666269302368, "reward_count_adherence_std": 0.0890870913863182, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.913884162902832, "reward_repeat_soft_std": 0.05960848182439804, "reward_judge_quality_mean": 0.2849999964237213, "reward_judge_quality_std": 0.12177261710166931, "reward_total_composite_mean": 0.4698163866996765, "reward_total_composite_std": 0.39093929529190063} {"timestamp_utc": "2026-04-13T01:28:03Z", "mode": "train", "global_step": 1272, "epoch": 0.12777498744349572, "loss": -0.0302, "grad_norm": 7.6495771408081055, "learning_rate": 6.148484848484849e-06, "num_tokens": 2280449.0, "completions/mean_length": 93.875, "completions/min_length": 72.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.875, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.9960002899169922, "rewards/meter/std": 0.0012340518878772855, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.823009729385376, "rewards/repeat_soft/std": 0.12556292116641998, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7971261143684387, "rewards/total_composite/std": 0.017181724309921265, "reward": 0.7971261143684387, "reward_std": 0.017181726172566414, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11837442219257355, "sampling/sampling_logp_difference/max": 1.977090835571289, "sampling/importance_sampling_ratio/min": 0.13847148418426514, "sampling/importance_sampling_ratio/mean": 1.0053144693374634, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.906210158020258, "clip_ratio/low_mean": 0.03368555149063468, "clip_ratio/low_min": 0.03368555149063468, "clip_ratio/high_mean": 0.06715391296893358, "clip_ratio/high_max": 0.06715391296893358, "clip_ratio/region_mean": 0.10083946445956826, "reward_total_mean": 0.7971261143684387, "reward_meter_mean": 0.9960002899169922, "reward_meter_std": 0.0012340518878772855, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.823009729385376, "reward_repeat_soft_std": 0.12556292116641998, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7971261143684387, "reward_total_composite_std": 0.017181724309921265} {"timestamp_utc": "2026-04-13T01:28:10Z", "mode": "train", "global_step": 1273, "epoch": 0.12787543947764943, "loss": 0.0432, "grad_norm": 11.235105514526367, "learning_rate": 6.1454545454545454e-06, "num_tokens": 2282321.0, "completions/mean_length": 62.0, "completions/min_length": 46.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.8536187410354614, "rewards/meter/std": 0.19678178429603577, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9453190565109253, "rewards/repeat_soft/std": 0.033207546919584274, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7167854309082031, "rewards/total_composite/std": 0.09179316461086273, "reward": 0.7167854309082031, "reward_std": 0.09179315716028214, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15855641663074493, "sampling/sampling_logp_difference/max": 1.4136056900024414, "sampling/importance_sampling_ratio/min": 0.2432645708322525, "sampling/importance_sampling_ratio/mean": 1.0218677520751953, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1834375858306885, "clip_ratio/low_mean": 0.05758236348628998, "clip_ratio/low_min": 0.05758236348628998, "clip_ratio/high_mean": 0.09173124004155397, "clip_ratio/high_max": 0.09173124004155397, "clip_ratio/region_mean": 0.14931360352784395, "reward_total_mean": 0.7167854309082031, "reward_meter_mean": 0.8536187410354614, "reward_meter_std": 0.19678178429603577, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9453190565109253, "reward_repeat_soft_std": 0.033207546919584274, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7167854309082031, "reward_total_composite_std": 0.09179316461086273} {"timestamp_utc": "2026-04-13T01:28:17Z", "mode": "train", "global_step": 1274, "epoch": 0.1279758915118031, "loss": 0.0468, "grad_norm": 17.923381805419922, "learning_rate": 6.142424242424243e-06, "num_tokens": 2283788.0, "completions/mean_length": 23.375, "completions/min_length": 20.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.375, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.760195255279541, "rewards/meter/std": 0.2100221961736679, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.7188378572463989, "rewards/total_composite/std": 0.0944327712059021, "reward": 0.7188378572463989, "reward_std": 0.0944327712059021, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1887337565422058, "sampling/sampling_logp_difference/max": 1.824930191040039, "sampling/importance_sampling_ratio/min": 0.16122891008853912, "sampling/importance_sampling_ratio/mean": 1.0310455560684204, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3815999776124954, "clip_ratio/low_mean": 0.08208333374932408, "clip_ratio/low_min": 0.08208333374932408, "clip_ratio/high_mean": 0.030193237122148275, "clip_ratio/high_max": 0.030193237122148275, "clip_ratio/region_mean": 0.11227657087147236, "reward_total_mean": 0.7188378572463989, "reward_meter_mean": 0.760195255279541, "reward_meter_std": 0.2100221961736679, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.7188378572463989, "reward_total_composite_std": 0.0944327712059021} {"timestamp_utc": "2026-04-13T01:28:23Z", "mode": "train", "global_step": 1275, "epoch": 0.12807634354595682, "loss": -0.0224, "grad_norm": 15.197134017944336, "learning_rate": 6.139393939393939e-06, "num_tokens": 2285668.0, "completions/mean_length": 55.0, "completions/min_length": 43.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.0, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9258548021316528, "rewards/meter/std": 0.17685122787952423, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9688565731048584, "rewards/repeat_soft/std": 0.03351173549890518, "rewards/judge_quality/mean": 0.7737500667572021, "rewards/judge_quality/std": 0.2260807603597641, "rewards/total_composite/mean": 0.895645260810852, "rewards/total_composite/std": 0.09070843458175659, "reward": 0.895645260810852, "reward_std": 0.0907084196805954, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12224767357110977, "sampling/sampling_logp_difference/max": 1.778864860534668, "sampling/importance_sampling_ratio/min": 0.16882969439029694, "sampling/importance_sampling_ratio/mean": 1.015145182609558, "sampling/importance_sampling_ratio/max": 1.8187263011932373, "entropy": 1.127294197678566, "clip_ratio/low_mean": 0.027579085202887654, "clip_ratio/low_min": 0.027579085202887654, "clip_ratio/high_mean": 0.058537607081234455, "clip_ratio/high_max": 0.058537607081234455, "clip_ratio/region_mean": 0.08611669228412211, "reward_total_mean": 0.895645260810852, "reward_meter_mean": 0.9258548021316528, "reward_meter_std": 0.17685122787952423, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9688565731048584, "reward_repeat_soft_std": 0.03351173549890518, "reward_judge_quality_mean": 0.7737500667572021, "reward_judge_quality_std": 0.2260807603597641, "reward_total_composite_mean": 0.895645260810852, "reward_total_composite_std": 0.09070843458175659} {"timestamp_utc": "2026-04-13T01:28:32Z", "mode": "train", "global_step": 1276, "epoch": 0.1281767955801105, "loss": -0.016, "grad_norm": 8.760083198547363, "learning_rate": 6.136363636363637e-06, "num_tokens": 2288194.0, "completions/mean_length": 107.75, "completions/min_length": 90.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.75, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.9885115027427673, "rewards/meter/std": 0.010799812152981758, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9405363202095032, "rewards/repeat_soft/std": 0.03051646798849106, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7897588014602661, "rewards/total_composite/std": 0.01986403949558735, "reward": 0.7897588014602661, "reward_std": 0.019864032045006752, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1428833156824112, "sampling/sampling_logp_difference/max": 1.882476806640625, "sampling/importance_sampling_ratio/min": 0.20934046804904938, "sampling/importance_sampling_ratio/mean": 1.033610224723816, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0983699932694435, "clip_ratio/low_mean": 0.06798850931227207, "clip_ratio/low_min": 0.06798850931227207, "clip_ratio/high_mean": 0.048417496494948864, "clip_ratio/high_max": 0.048417496494948864, "clip_ratio/region_mean": 0.11640600580722094, "reward_total_mean": 0.7897588014602661, "reward_meter_mean": 0.9885115027427673, "reward_meter_std": 0.010799812152981758, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9405363202095032, "reward_repeat_soft_std": 0.03051646798849106, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7897588014602661, "reward_total_composite_std": 0.01986403949558735} {"timestamp_utc": "2026-04-13T01:28:38Z", "mode": "train", "global_step": 1277, "epoch": 0.12827724761426418, "loss": 0.0371, "grad_norm": 17.716920852661133, "learning_rate": 6.133333333333334e-06, "num_tokens": 2289678.0, "completions/mean_length": 36.5, "completions/min_length": 33.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.5, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.7588077783584595, "rewards/meter/std": 0.28657376766204834, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9959182739257812, "rewards/repeat_soft/std": 0.006008748430758715, "rewards/judge_quality/mean": 0.5387499928474426, "rewards/judge_quality/std": 0.24485784769058228, "rewards/total_composite/mean": 0.7526803016662598, "rewards/total_composite/std": 0.14789964258670807, "reward": 0.7526803016662598, "reward_std": 0.14789965748786926, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.143199160695076, "sampling/sampling_logp_difference/max": 1.7191977500915527, "sampling/importance_sampling_ratio/min": 0.1792098581790924, "sampling/importance_sampling_ratio/mean": 1.0125808715820312, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6421382464468479, "clip_ratio/low_mean": 0.03867469076067209, "clip_ratio/low_min": 0.03867469076067209, "clip_ratio/high_mean": 0.07826426159590483, "clip_ratio/high_max": 0.07826426159590483, "clip_ratio/region_mean": 0.11693895235657692, "reward_total_mean": 0.7526803016662598, "reward_meter_mean": 0.7588077783584595, "reward_meter_std": 0.28657376766204834, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9959182739257812, "reward_repeat_soft_std": 0.006008748430758715, "reward_judge_quality_mean": 0.5387499928474426, "reward_judge_quality_std": 0.24485784769058228, "reward_total_composite_mean": 0.7526803016662598, "reward_total_composite_std": 0.14789964258670807} {"timestamp_utc": "2026-04-13T01:28:50Z", "mode": "train", "global_step": 1278, "epoch": 0.1283776996484179, "loss": -0.1164, "grad_norm": 2.154184103012085, "learning_rate": 6.130303030303031e-06, "num_tokens": 2291186.0, "completions/mean_length": 97.5, "completions/min_length": 33.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 38.28571701049805, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.9768309593200684, "rewards/meter/std": 0.03893294185400009, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9963254928588867, "rewards/repeat_soft/std": 0.005535479169338942, "rewards/judge_quality/mean": 0.4125000238418579, "rewards/judge_quality/std": 0.19557608664035797, "rewards/total_composite/mean": 0.7259897589683533, "rewards/total_composite/std": 0.29580792784690857, "reward": 0.7259897589683533, "reward_std": 0.29580792784690857, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15815457701683044, "sampling/sampling_logp_difference/max": 0.955815315246582, "sampling/importance_sampling_ratio/min": 0.3844985365867615, "sampling/importance_sampling_ratio/mean": 1.0273667573928833, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9712617918848991, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10234853532165289, "clip_ratio/high_max": 0.10234853532165289, "clip_ratio/region_mean": 0.10234853532165289, "reward_total_mean": 0.7259897589683533, "reward_meter_mean": 0.9768309593200684, "reward_meter_std": 0.03893294185400009, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9963254928588867, "reward_repeat_soft_std": 0.005535479169338942, "reward_judge_quality_mean": 0.4125000238418579, "reward_judge_quality_std": 0.19557608664035797, "reward_total_composite_mean": 0.7259897589683533, "reward_total_composite_std": 0.29580792784690857} {"timestamp_utc": "2026-04-13T01:28:57Z", "mode": "train", "global_step": 1279, "epoch": 0.12847815168257157, "loss": 0.0554, "grad_norm": 15.590206146240234, "learning_rate": 6.127272727272727e-06, "num_tokens": 2292882.0, "completions/mean_length": 40.0, "completions/min_length": 35.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.9073842763900757, "rewards/meter/std": 0.15413229167461395, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9633268117904663, "rewards/repeat_soft/std": 0.021767502650618553, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.7851555943489075, "rewards/total_composite/std": 0.07088543474674225, "reward": 0.7851555943489075, "reward_std": 0.07088542729616165, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13574829697608948, "sampling/sampling_logp_difference/max": 1.5556225776672363, "sampling/importance_sampling_ratio/min": 0.21105793118476868, "sampling/importance_sampling_ratio/mean": 1.005776286125183, "sampling/importance_sampling_ratio/max": 1.8886284828186035, "entropy": 0.8968902677297592, "clip_ratio/low_mean": 0.011627906933426857, "clip_ratio/low_min": 0.011627906933426857, "clip_ratio/high_mean": 0.13679059874266386, "clip_ratio/high_max": 0.13679059874266386, "clip_ratio/region_mean": 0.14841850567609072, "reward_total_mean": 0.7851555943489075, "reward_meter_mean": 0.9073842763900757, "reward_meter_std": 0.15413229167461395, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9633268117904663, "reward_repeat_soft_std": 0.021767502650618553, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.7851555943489075, "reward_total_composite_std": 0.07088543474674225} {"timestamp_utc": "2026-04-13T01:29:04Z", "mode": "train", "global_step": 1280, "epoch": 0.12857860371672528, "loss": 0.0134, "grad_norm": 18.654632568359375, "learning_rate": 6.1242424242424245e-06, "num_tokens": 2294536.0, "completions/mean_length": 30.75, "completions/min_length": 26.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.75, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.8905718326568604, "rewards/meter/std": 0.1395193189382553, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8989852666854858, "rewards/repeat_soft/std": 0.09635616838932037, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.7865308523178101, "rewards/total_composite/std": 0.05268998071551323, "reward": 0.7865308523178101, "reward_std": 0.05268998444080353, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07430548965930939, "sampling/sampling_logp_difference/max": 1.580556869506836, "sampling/importance_sampling_ratio/min": 0.2058604210615158, "sampling/importance_sampling_ratio/mean": 1.0026893615722656, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4177478849887848, "clip_ratio/low_mean": 0.028494623489677906, "clip_ratio/low_min": 0.028494623489677906, "clip_ratio/high_mean": 0.05788620188832283, "clip_ratio/high_max": 0.05788620188832283, "clip_ratio/region_mean": 0.08638082537800074, "reward_total_mean": 0.7865308523178101, "reward_meter_mean": 0.8905718326568604, "reward_meter_std": 0.1395193189382553, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8989852666854858, "reward_repeat_soft_std": 0.09635616838932037, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.7865308523178101, "reward_total_composite_std": 0.05268998071551323} {"timestamp_utc": "2026-04-13T01:29:16Z", "mode": "train", "global_step": 1281, "epoch": 0.12867905575087896, "loss": -0.0476, "grad_norm": 6.230748653411865, "learning_rate": 6.121212121212121e-06, "num_tokens": 2296508.0, "completions/mean_length": 153.5, "completions/min_length": 87.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 102.28572082519531, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.6169472932815552, "rewards/meter/std": 0.3613012433052063, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.2828427255153656, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9569189548492432, "rewards/repeat_soft/std": 0.06951666623353958, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.6204432249069214, "rewards/total_composite/std": 0.15148240327835083, "reward": 0.6204432249069214, "reward_std": 0.15148241817951202, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17205782234668732, "sampling/sampling_logp_difference/max": 1.8611621856689453, "sampling/importance_sampling_ratio/min": 0.15549181401729584, "sampling/importance_sampling_ratio/mean": 1.0151416063308716, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1544170156121254, "clip_ratio/low_mean": 0.060983140021562576, "clip_ratio/low_min": 0.060983140021562576, "clip_ratio/high_mean": 0.07509571220725775, "clip_ratio/high_max": 0.07509571220725775, "clip_ratio/region_mean": 0.13607885222882032, "reward_total_mean": 0.6204432249069214, "reward_meter_mean": 0.6169472932815552, "reward_meter_std": 0.3613012433052063, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.2828427255153656, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9569189548492432, "reward_repeat_soft_std": 0.06951666623353958, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.6204432249069214, "reward_total_composite_std": 0.15148240327835083} {"timestamp_utc": "2026-04-13T01:29:22Z", "mode": "train", "global_step": 1282, "epoch": 0.12877950778503264, "loss": -0.0036, "grad_norm": 20.53963851928711, "learning_rate": 6.118181818181819e-06, "num_tokens": 2297882.0, "completions/mean_length": 19.75, "completions/min_length": 17.0, "completions/max_length": 22.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.75, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 22.0, "rewards/meter/mean": 0.9823153018951416, "rewards/meter/std": 0.010364190675318241, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.942566454410553, "rewards/repeat_soft/std": 0.05056637525558472, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.8167985081672668, "rewards/total_composite/std": 0.00769049534574151, "reward": 0.8167985081672668, "reward_std": 0.007690491154789925, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17215292155742645, "sampling/sampling_logp_difference/max": 1.5083866119384766, "sampling/importance_sampling_ratio/min": 0.22126668691635132, "sampling/importance_sampling_ratio/mean": 1.0289076566696167, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3936190828680992, "clip_ratio/low_mean": 0.08432662533596158, "clip_ratio/low_min": 0.08432662533596158, "clip_ratio/high_mean": 0.11571067944169044, "clip_ratio/high_max": 0.11571067944169044, "clip_ratio/region_mean": 0.20003730477765203, "reward_total_mean": 0.8167985081672668, "reward_meter_mean": 0.9823153018951416, "reward_meter_std": 0.010364190675318241, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.942566454410553, "reward_repeat_soft_std": 0.05056637525558472, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.8167985081672668, "reward_total_composite_std": 0.00769049534574151} {"timestamp_utc": "2026-04-13T01:29:34Z", "mode": "train", "global_step": 1283, "epoch": 0.12887995981918635, "loss": -0.1175, "grad_norm": 1.7600758075714111, "learning_rate": 6.115151515151516e-06, "num_tokens": 2299492.0, "completions/mean_length": 97.25, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 38.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.8338295221328735, "rewards/meter/std": 0.3206214904785156, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9245314598083496, "rewards/repeat_soft/std": 0.10322266817092896, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.6871087551116943, "rewards/total_composite/std": 0.27979540824890137, "reward": 0.6871087551116943, "reward_std": 0.27979540824890137, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16665180027484894, "sampling/sampling_logp_difference/max": 2.253872871398926, "sampling/importance_sampling_ratio/min": 0.10499181598424911, "sampling/importance_sampling_ratio/mean": 1.018015742301941, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0429059192538261, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.11460980772972107, "clip_ratio/high_max": 0.11460980772972107, "clip_ratio/region_mean": 0.11460980772972107, "reward_total_mean": 0.6871087551116943, "reward_meter_mean": 0.8338295221328735, "reward_meter_std": 0.3206214904785156, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9245314598083496, "reward_repeat_soft_std": 0.10322266817092896, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.6871087551116943, "reward_total_composite_std": 0.27979540824890137} {"timestamp_utc": "2026-04-13T01:29:41Z", "mode": "train", "global_step": 1284, "epoch": 0.12898041185334003, "loss": 0.0328, "grad_norm": 11.730399131774902, "learning_rate": 6.112121212121213e-06, "num_tokens": 2301566.0, "completions/mean_length": 73.25, "completions/min_length": 58.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.25, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.7769885659217834, "rewards/meter/std": 0.38416457176208496, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9087756872177124, "rewards/repeat_soft/std": 0.10984205454587936, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.19949938356876373, "rewards/total_composite/mean": 0.7084599733352661, "rewards/total_composite/std": 0.17839927971363068, "reward": 0.7084599733352661, "reward_std": 0.17839929461479187, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.132815420627594, "sampling/sampling_logp_difference/max": 1.790284514427185, "sampling/importance_sampling_ratio/min": 0.16691267490386963, "sampling/importance_sampling_ratio/mean": 1.0007115602493286, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9798158183693886, "clip_ratio/low_mean": 0.03565041348338127, "clip_ratio/low_min": 0.03565041348338127, "clip_ratio/high_mean": 0.0968780005350709, "clip_ratio/high_max": 0.0968780005350709, "clip_ratio/region_mean": 0.13252841401845217, "reward_total_mean": 0.7084599733352661, "reward_meter_mean": 0.7769885659217834, "reward_meter_std": 0.38416457176208496, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9087756872177124, "reward_repeat_soft_std": 0.10984205454587936, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.19949938356876373, "reward_total_composite_mean": 0.7084599733352661, "reward_total_composite_std": 0.17839927971363068} {"timestamp_utc": "2026-04-13T01:29:49Z", "mode": "train", "global_step": 1285, "epoch": 0.12908086388749374, "loss": -0.0194, "grad_norm": 9.210697174072266, "learning_rate": 6.10909090909091e-06, "num_tokens": 2303710.0, "completions/mean_length": 69.0, "completions/min_length": 65.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9185985922813416, "rewards/meter/std": 0.16910193860530853, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.979729413986206, "rewards/repeat_soft/std": 0.016979997977614403, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.7985922694206238, "rewards/total_composite/std": 0.0874263122677803, "reward": 0.7985922694206238, "reward_std": 0.0874263197183609, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13370870053768158, "sampling/sampling_logp_difference/max": 1.9505109786987305, "sampling/importance_sampling_ratio/min": 0.14220137894153595, "sampling/importance_sampling_ratio/mean": 1.0214465856552124, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8661224097013474, "clip_ratio/low_mean": 0.03560520429164171, "clip_ratio/low_min": 0.03560520429164171, "clip_ratio/high_mean": 0.07541296072304249, "clip_ratio/high_max": 0.07541296072304249, "clip_ratio/region_mean": 0.1110181650146842, "reward_total_mean": 0.7985922694206238, "reward_meter_mean": 0.9185985922813416, "reward_meter_std": 0.16910193860530853, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.979729413986206, "reward_repeat_soft_std": 0.016979997977614403, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.7985922694206238, "reward_total_composite_std": 0.0874263122677803} {"timestamp_utc": "2026-04-13T01:29:56Z", "mode": "train", "global_step": 1286, "epoch": 0.12918131592164742, "loss": 0.0634, "grad_norm": 18.006166458129883, "learning_rate": 6.106060606060606e-06, "num_tokens": 2305322.0, "completions/mean_length": 43.5, "completions/min_length": 36.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.781651496887207, "rewards/meter/std": 0.33994168043136597, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.943294882774353, "rewards/repeat_soft/std": 0.03565758466720581, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7156976461410522, "rewards/total_composite/std": 0.15153686702251434, "reward": 0.7156976461410522, "reward_std": 0.15153685212135315, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1679685264825821, "sampling/sampling_logp_difference/max": 1.8383724689483643, "sampling/importance_sampling_ratio/min": 0.15907612442970276, "sampling/importance_sampling_ratio/mean": 1.0006334781646729, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9998492896556854, "clip_ratio/low_mean": 0.037145345471799374, "clip_ratio/low_min": 0.037145345471799374, "clip_ratio/high_mean": 0.14621571078896523, "clip_ratio/high_max": 0.14621571078896523, "clip_ratio/region_mean": 0.1833610562607646, "reward_total_mean": 0.7156976461410522, "reward_meter_mean": 0.781651496887207, "reward_meter_std": 0.33994168043136597, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.943294882774353, "reward_repeat_soft_std": 0.03565758466720581, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7156976461410522, "reward_total_composite_std": 0.15153686702251434} {"timestamp_utc": "2026-04-13T01:30:03Z", "mode": "train", "global_step": 1287, "epoch": 0.1292817679558011, "loss": -0.0027, "grad_norm": 12.27800464630127, "learning_rate": 6.103030303030304e-06, "num_tokens": 2306930.0, "completions/mean_length": 47.0, "completions/min_length": 37.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.8145810961723328, "rewards/meter/std": 0.214928537607193, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.983119010925293, "rewards/repeat_soft/std": 0.006984120700508356, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.7896233797073364, "rewards/total_composite/std": 0.1340511292219162, "reward": 0.7896233797073364, "reward_std": 0.1340511292219162, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14933830499649048, "sampling/sampling_logp_difference/max": 1.9547739028930664, "sampling/importance_sampling_ratio/min": 0.1415964812040329, "sampling/importance_sampling_ratio/mean": 1.0095773935317993, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7334991842508316, "clip_ratio/low_mean": 0.09168575145304203, "clip_ratio/low_min": 0.09168575145304203, "clip_ratio/high_mean": 0.0678683165460825, "clip_ratio/high_max": 0.0678683165460825, "clip_ratio/region_mean": 0.15955406799912453, "reward_total_mean": 0.7896233797073364, "reward_meter_mean": 0.8145810961723328, "reward_meter_std": 0.214928537607193, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.983119010925293, "reward_repeat_soft_std": 0.006984120700508356, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.7896233797073364, "reward_total_composite_std": 0.1340511292219162} {"timestamp_utc": "2026-04-13T01:30:09Z", "mode": "train", "global_step": 1288, "epoch": 0.1293822199899548, "loss": 0.0398, "grad_norm": 16.800533294677734, "learning_rate": 6.1e-06, "num_tokens": 2308523.0, "completions/mean_length": 26.125, "completions/min_length": 22.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.125, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.7667789459228516, "rewards/meter/std": 0.3774774968624115, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9271124601364136, "rewards/repeat_soft/std": 0.04947630688548088, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.7107617259025574, "rewards/total_composite/std": 0.1758957952260971, "reward": 0.7107617259025574, "reward_std": 0.1758957952260971, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12878061830997467, "sampling/sampling_logp_difference/max": 1.3348965644836426, "sampling/importance_sampling_ratio/min": 0.26318538188934326, "sampling/importance_sampling_ratio/mean": 1.0074741840362549, "sampling/importance_sampling_ratio/max": 1.8642228841781616, "entropy": 1.0040631853044033, "clip_ratio/low_mean": 0.02314814832061529, "clip_ratio/low_min": 0.02314814832061529, "clip_ratio/high_mean": 0.11754899192601442, "clip_ratio/high_max": 0.11754899192601442, "clip_ratio/region_mean": 0.14069714024662971, "reward_total_mean": 0.7107617259025574, "reward_meter_mean": 0.7667789459228516, "reward_meter_std": 0.3774774968624115, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9271124601364136, "reward_repeat_soft_std": 0.04947630688548088, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.7107617259025574, "reward_total_composite_std": 0.1758957952260971} {"timestamp_utc": "2026-04-13T01:30:16Z", "mode": "train", "global_step": 1289, "epoch": 0.12948267202410849, "loss": -0.0041, "grad_norm": 15.808647155761719, "learning_rate": 6.096969696969698e-06, "num_tokens": 2310387.0, "completions/mean_length": 49.0, "completions/min_length": 44.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.0, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.852415919303894, "rewards/meter/std": 0.19287897646427155, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9938690662384033, "rewards/repeat_soft/std": 0.0075904494151473045, "rewards/judge_quality/mean": 0.5099999904632568, "rewards/judge_quality/std": 0.2093527615070343, "rewards/total_composite/mean": 0.6904963254928589, "rewards/total_composite/std": 0.30063143372535706, "reward": 0.6904963254928589, "reward_std": 0.30063143372535706, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17714610695838928, "sampling/sampling_logp_difference/max": 3.787367820739746, "sampling/importance_sampling_ratio/min": 0.022655155509710312, "sampling/importance_sampling_ratio/mean": 1.0066498517990112, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7661812454462051, "clip_ratio/low_mean": 0.05174911022186279, "clip_ratio/low_min": 0.05174911022186279, "clip_ratio/high_mean": 0.10668369475752115, "clip_ratio/high_max": 0.10668369475752115, "clip_ratio/region_mean": 0.15843280497938395, "reward_total_mean": 0.6904963254928589, "reward_meter_mean": 0.852415919303894, "reward_meter_std": 0.19287897646427155, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9938690662384033, "reward_repeat_soft_std": 0.0075904494151473045, "reward_judge_quality_mean": 0.5099999904632568, "reward_judge_quality_std": 0.2093527615070343, "reward_total_composite_mean": 0.6904963254928589, "reward_total_composite_std": 0.30063143372535706} {"timestamp_utc": "2026-04-13T01:30:23Z", "mode": "train", "global_step": 1290, "epoch": 0.12958312405826217, "loss": -0.0203, "grad_norm": 9.47526741027832, "learning_rate": 6.0939393939393946e-06, "num_tokens": 2312145.0, "completions/mean_length": 59.75, "completions/min_length": 56.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.75, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9875797629356384, "rewards/meter/std": 0.005216935183852911, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.991989254951477, "rewards/repeat_soft/std": 0.010632848367094994, "rewards/judge_quality/mean": 0.5074999332427979, "rewards/judge_quality/std": 0.12464235723018646, "rewards/total_composite/mean": 0.8458598256111145, "rewards/total_composite/std": 0.03922465443611145, "reward": 0.8458598256111145, "reward_std": 0.03922465443611145, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15073354542255402, "sampling/sampling_logp_difference/max": 1.7370033264160156, "sampling/importance_sampling_ratio/min": 0.1760471761226654, "sampling/importance_sampling_ratio/mean": 1.0218689441680908, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.131323479115963, "clip_ratio/low_mean": 0.08141327789053321, "clip_ratio/low_min": 0.08141327789053321, "clip_ratio/high_mean": 0.052079553715884686, "clip_ratio/high_max": 0.052079553715884686, "clip_ratio/region_mean": 0.1334928316064179, "reward_total_mean": 0.8458598256111145, "reward_meter_mean": 0.9875797629356384, "reward_meter_std": 0.005216935183852911, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.991989254951477, "reward_repeat_soft_std": 0.010632848367094994, "reward_judge_quality_mean": 0.5074999332427979, "reward_judge_quality_std": 0.12464235723018646, "reward_total_composite_mean": 0.8458598256111145, "reward_total_composite_std": 0.03922465443611145} {"timestamp_utc": "2026-04-13T01:30:35Z", "mode": "train", "global_step": 1291, "epoch": 0.12968357609241588, "loss": -0.1157, "grad_norm": 3.2828962802886963, "learning_rate": 6.090909090909092e-06, "num_tokens": 2314235.0, "completions/mean_length": 132.25, "completions/min_length": 62.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 78.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.7115010023117065, "rewards/meter/std": 0.3635506331920624, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.14880475401878357, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8048301339149475, "rewards/repeat_soft/std": 0.07230329513549805, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.1345893144607544, "rewards/total_composite/mean": 0.581871747970581, "rewards/total_composite/std": 0.2768004536628723, "reward": 0.581871747970581, "reward_std": 0.2768004536628723, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1443389654159546, "sampling/sampling_logp_difference/max": 9.187141418457031, "sampling/importance_sampling_ratio/min": 0.00010234701767330989, "sampling/importance_sampling_ratio/mean": 0.9969649910926819, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6933770924806595, "clip_ratio/low_mean": 0.02185097523033619, "clip_ratio/low_min": 0.02185097523033619, "clip_ratio/high_mean": 0.08013410400599241, "clip_ratio/high_max": 0.08013410400599241, "clip_ratio/region_mean": 0.1019850792363286, "reward_total_mean": 0.581871747970581, "reward_meter_mean": 0.7115010023117065, "reward_meter_std": 0.3635506331920624, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.14880475401878357, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8048301339149475, "reward_repeat_soft_std": 0.07230329513549805, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.1345893144607544, "reward_total_composite_mean": 0.581871747970581, "reward_total_composite_std": 0.2768004536628723} {"timestamp_utc": "2026-04-13T01:30:47Z", "mode": "train", "global_step": 1292, "epoch": 0.12978402812656956, "loss": -0.153, "grad_norm": 2.8754000663757324, "learning_rate": 6.087878787878788e-06, "num_tokens": 2316004.0, "completions/mean_length": 119.125, "completions/min_length": 56.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 63.000003814697266, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.8843054175376892, "rewards/meter/std": 0.19714350998401642, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9805084466934204, "rewards/repeat_soft/std": 0.015506435185670853, "rewards/judge_quality/mean": 0.44875001907348633, "rewards/judge_quality/std": 0.2105392962694168, "rewards/total_composite/mean": 0.6915193796157837, "rewards/total_composite/std": 0.2982116937637329, "reward": 0.6915193796157837, "reward_std": 0.2982116937637329, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14273476600646973, "sampling/sampling_logp_difference/max": 1.4922869205474854, "sampling/importance_sampling_ratio/min": 0.2248578518629074, "sampling/importance_sampling_ratio/mean": 1.0128971338272095, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7384102493524551, "clip_ratio/low_mean": 0.02008928544819355, "clip_ratio/low_min": 0.02008928544819355, "clip_ratio/high_mean": 0.08724083937704563, "clip_ratio/high_max": 0.08724083937704563, "clip_ratio/region_mean": 0.10733012482523918, "reward_total_mean": 0.6915193796157837, "reward_meter_mean": 0.8843054175376892, "reward_meter_std": 0.19714350998401642, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.2651650309562683, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9805084466934204, "reward_repeat_soft_std": 0.015506435185670853, "reward_judge_quality_mean": 0.44875001907348633, "reward_judge_quality_std": 0.2105392962694168, "reward_total_composite_mean": 0.6915193796157837, "reward_total_composite_std": 0.2982116937637329} {"timestamp_utc": "2026-04-13T01:30:58Z", "mode": "train", "global_step": 1293, "epoch": 0.12988448016072326, "loss": -0.1089, "grad_norm": 1.819399356842041, "learning_rate": 6.0848484848484855e-06, "num_tokens": 2317598.0, "completions/mean_length": 228.25, "completions/min_length": 50.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.5098286271095276, "rewards/meter/std": 0.4463561475276947, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.5175492167472839, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9292809963226318, "rewards/repeat_soft/std": 0.08274932950735092, "rewards/judge_quality/mean": 0.28125, "rewards/judge_quality/std": 0.1914931833744049, "rewards/total_composite/mean": 0.4573509693145752, "rewards/total_composite/std": 0.3849949240684509, "reward": 0.4573509693145752, "reward_std": 0.3849949240684509, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19183403253555298, "sampling/sampling_logp_difference/max": 6.282299995422363, "sampling/importance_sampling_ratio/min": 0.001869096769951284, "sampling/importance_sampling_ratio/mean": 1.0381250381469727, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8039544150233269, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09624598175287247, "clip_ratio/high_max": 0.09624598175287247, "clip_ratio/region_mean": 0.09624598175287247, "reward_total_mean": 0.4573509693145752, "reward_meter_mean": 0.5098286271095276, "reward_meter_std": 0.4463561475276947, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.5175492167472839, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9292809963226318, "reward_repeat_soft_std": 0.08274932950735092, "reward_judge_quality_mean": 0.28125, "reward_judge_quality_std": 0.1914931833744049, "reward_total_composite_mean": 0.4573509693145752, "reward_total_composite_std": 0.3849949240684509} {"timestamp_utc": "2026-04-13T01:31:06Z", "mode": "train", "global_step": 1294, "epoch": 0.12998493219487695, "loss": 0.0684, "grad_norm": 33.61226272583008, "learning_rate": 6.081818181818182e-06, "num_tokens": 2318961.0, "completions/mean_length": 20.375, "completions/min_length": 17.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.375, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.8890545964241028, "rewards/meter/std": 0.2749333083629608, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9563953280448914, "rewards/repeat_soft/std": 0.013796630315482616, "rewards/judge_quality/mean": 0.45249998569488525, "rewards/judge_quality/std": 0.21224987506866455, "rewards/total_composite/mean": 0.7814640998840332, "rewards/total_composite/std": 0.1390138566493988, "reward": 0.7814640998840332, "reward_std": 0.1390138417482376, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1853715181350708, "sampling/sampling_logp_difference/max": 1.6152563095092773, "sampling/importance_sampling_ratio/min": 0.1988396942615509, "sampling/importance_sampling_ratio/mean": 0.9962299466133118, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1683172583580017, "clip_ratio/low_mean": 0.0052083334885537624, "clip_ratio/low_min": 0.0052083334885537624, "clip_ratio/high_mean": 0.1652777809649706, "clip_ratio/high_max": 0.1652777809649706, "clip_ratio/region_mean": 0.17048611445352435, "reward_total_mean": 0.7814640998840332, "reward_meter_mean": 0.8890545964241028, "reward_meter_std": 0.2749333083629608, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9563953280448914, "reward_repeat_soft_std": 0.013796630315482616, "reward_judge_quality_mean": 0.45249998569488525, "reward_judge_quality_std": 0.21224987506866455, "reward_total_composite_mean": 0.7814640998840332, "reward_total_composite_std": 0.1390138566493988} {"timestamp_utc": "2026-04-13T01:31:14Z", "mode": "train", "global_step": 1295, "epoch": 0.13008538422903063, "loss": -0.0055, "grad_norm": 6.736032485961914, "learning_rate": 6.07878787878788e-06, "num_tokens": 2321395.0, "completions/mean_length": 117.25, "completions/min_length": 108.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.25, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.9262022972106934, "rewards/meter/std": 0.1449710875749588, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7328628301620483, "rewards/repeat_soft/std": 0.08564739674329758, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7469522953033447, "rewards/total_composite/std": 0.06264089792966843, "reward": 0.7469522953033447, "reward_std": 0.06264090538024902, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10160365700721741, "sampling/sampling_logp_difference/max": 3.1599650382995605, "sampling/importance_sampling_ratio/min": 0.042427223175764084, "sampling/importance_sampling_ratio/mean": 1.0064301490783691, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6837055459618568, "clip_ratio/low_mean": 0.052067878656089306, "clip_ratio/low_min": 0.052067878656089306, "clip_ratio/high_mean": 0.05182449147105217, "clip_ratio/high_max": 0.05182449147105217, "clip_ratio/region_mean": 0.10389237012714148, "reward_total_mean": 0.7469522953033447, "reward_meter_mean": 0.9262022972106934, "reward_meter_std": 0.1449710875749588, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7328628301620483, "reward_repeat_soft_std": 0.08564739674329758, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7469522953033447, "reward_total_composite_std": 0.06264089792966843} {"timestamp_utc": "2026-04-13T01:31:23Z", "mode": "train", "global_step": 1296, "epoch": 0.13018583626318433, "loss": 0.0081, "grad_norm": 14.727508544921875, "learning_rate": 6.0757575757575755e-06, "num_tokens": 2322985.0, "completions/mean_length": 45.75, "completions/min_length": 38.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.75, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.8682267665863037, "rewards/meter/std": 0.3021674156188965, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9767292737960815, "rewards/repeat_soft/std": 0.018223747611045837, "rewards/judge_quality/mean": 0.5012500286102295, "rewards/judge_quality/std": 0.16974246501922607, "rewards/total_composite/mean": 0.7887499332427979, "rewards/total_composite/std": 0.15186181664466858, "reward": 0.7887499332427979, "reward_std": 0.15186183154582977, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16035249829292297, "sampling/sampling_logp_difference/max": 1.980534553527832, "sampling/importance_sampling_ratio/min": 0.13799545168876648, "sampling/importance_sampling_ratio/mean": 1.0315289497375488, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1178889572620392, "clip_ratio/low_mean": 0.013586956076323986, "clip_ratio/low_min": 0.013586956076323986, "clip_ratio/high_mean": 0.1397784696891904, "clip_ratio/high_max": 0.1397784696891904, "clip_ratio/region_mean": 0.15336542576551437, "reward_total_mean": 0.7887499332427979, "reward_meter_mean": 0.8682267665863037, "reward_meter_std": 0.3021674156188965, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9767292737960815, "reward_repeat_soft_std": 0.018223747611045837, "reward_judge_quality_mean": 0.5012500286102295, "reward_judge_quality_std": 0.16974246501922607, "reward_total_composite_mean": 0.7887499332427979, "reward_total_composite_std": 0.15186181664466858} {"timestamp_utc": "2026-04-13T01:31:30Z", "mode": "train", "global_step": 1297, "epoch": 0.13028628829733802, "loss": 0.0051, "grad_norm": 16.704870223999023, "learning_rate": 6.072727272727274e-06, "num_tokens": 2324478.0, "completions/mean_length": 37.625, "completions/min_length": 32.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.625, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.757331132888794, "rewards/meter/std": 0.42793774604797363, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9740206003189087, "rewards/repeat_soft/std": 0.026715973392128944, "rewards/judge_quality/mean": 0.6599999666213989, "rewards/judge_quality/std": 0.23384979367256165, "rewards/total_composite/mean": 0.7862011194229126, "rewards/total_composite/std": 0.20006966590881348, "reward": 0.7862011194229126, "reward_std": 0.20006965100765228, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17066019773483276, "sampling/sampling_logp_difference/max": 1.3763411045074463, "sampling/importance_sampling_ratio/min": 0.2525007426738739, "sampling/importance_sampling_ratio/mean": 1.020906925201416, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0895232111215591, "clip_ratio/low_mean": 0.05220588389784098, "clip_ratio/low_min": 0.05220588389784098, "clip_ratio/high_mean": 0.1168175507336855, "clip_ratio/high_max": 0.1168175507336855, "clip_ratio/region_mean": 0.16902343463152647, "reward_total_mean": 0.7862011194229126, "reward_meter_mean": 0.757331132888794, "reward_meter_std": 0.42793774604797363, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9740206003189087, "reward_repeat_soft_std": 0.026715973392128944, "reward_judge_quality_mean": 0.6599999666213989, "reward_judge_quality_std": 0.23384979367256165, "reward_total_composite_mean": 0.7862011194229126, "reward_total_composite_std": 0.20006966590881348} {"timestamp_utc": "2026-04-13T01:31:42Z", "mode": "train", "global_step": 1298, "epoch": 0.13038674033149172, "loss": -0.1639, "grad_norm": 2.5447590351104736, "learning_rate": 6.06969696969697e-06, "num_tokens": 2326397.0, "completions/mean_length": 129.875, "completions/min_length": 62.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 75.28572082519531, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.7545531392097473, "rewards/meter/std": 0.36815041303634644, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9371895790100098, "rewards/repeat_soft/std": 0.037625838071107864, "rewards/judge_quality/mean": 0.3399999737739563, "rewards/judge_quality/std": 0.15052290260791779, "rewards/total_composite/mean": 0.6502236723899841, "rewards/total_composite/std": 0.28095555305480957, "reward": 0.6502236723899841, "reward_std": 0.28095558285713196, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1641642451286316, "sampling/sampling_logp_difference/max": 1.887235403060913, "sampling/importance_sampling_ratio/min": 0.15149003267288208, "sampling/importance_sampling_ratio/mean": 1.0144641399383545, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0057775303721428, "clip_ratio/low_mean": 0.020220588892698288, "clip_ratio/low_min": 0.020220588892698288, "clip_ratio/high_mean": 0.13792523369193077, "clip_ratio/high_max": 0.13792523369193077, "clip_ratio/region_mean": 0.15814582258462906, "reward_total_mean": 0.6502236723899841, "reward_meter_mean": 0.7545531392097473, "reward_meter_std": 0.36815041303634644, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9371895790100098, "reward_repeat_soft_std": 0.037625838071107864, "reward_judge_quality_mean": 0.3399999737739563, "reward_judge_quality_std": 0.15052290260791779, "reward_total_composite_mean": 0.6502236723899841, "reward_total_composite_std": 0.28095555305480957} {"timestamp_utc": "2026-04-13T01:31:52Z", "mode": "train", "global_step": 1299, "epoch": 0.1304871923656454, "loss": -0.0313, "grad_norm": 14.94461441040039, "learning_rate": 6.066666666666667e-06, "num_tokens": 2328062.0, "completions/mean_length": 37.125, "completions/min_length": 31.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.125, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.5722876787185669, "rewards/meter/std": 0.445983350276947, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9815419912338257, "rewards/repeat_soft/std": 0.025572631508111954, "rewards/judge_quality/mean": 0.38874998688697815, "rewards/judge_quality/std": 0.08675704896450043, "rewards/total_composite/mean": 0.622308611869812, "rewards/total_composite/std": 0.20226947963237762, "reward": 0.622308611869812, "reward_std": 0.20226947963237762, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1501959264278412, "sampling/sampling_logp_difference/max": 1.760427474975586, "sampling/importance_sampling_ratio/min": 0.17197133600711823, "sampling/importance_sampling_ratio/mean": 1.0064576864242554, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1033459082245827, "clip_ratio/low_mean": 0.09344859980046749, "clip_ratio/low_min": 0.09344859980046749, "clip_ratio/high_mean": 0.0960961151868105, "clip_ratio/high_max": 0.0960961151868105, "clip_ratio/region_mean": 0.18954471498727798, "reward_total_mean": 0.622308611869812, "reward_meter_mean": 0.5722876787185669, "reward_meter_std": 0.445983350276947, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9815419912338257, "reward_repeat_soft_std": 0.025572631508111954, "reward_judge_quality_mean": 0.38874998688697815, "reward_judge_quality_std": 0.08675704896450043, "reward_total_composite_mean": 0.622308611869812, "reward_total_composite_std": 0.20226947963237762} {"timestamp_utc": "2026-04-13T01:31:59Z", "mode": "train", "global_step": 1300, "epoch": 0.13058764439979909, "loss": -0.005, "grad_norm": 10.266416549682617, "learning_rate": 6.063636363636364e-06, "num_tokens": 2330119.0, "completions/mean_length": 66.125, "completions/min_length": 58.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.125, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.8918511271476746, "rewards/meter/std": 0.2632952928543091, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8825685977935791, "rewards/repeat_soft/std": 0.09118378907442093, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7528398633003235, "rewards/total_composite/std": 0.1126292496919632, "reward": 0.7528398633003235, "reward_std": 0.1126292422413826, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1068713366985321, "sampling/sampling_logp_difference/max": 1.3770532608032227, "sampling/importance_sampling_ratio/min": 0.2523209750652313, "sampling/importance_sampling_ratio/mean": 1.0144987106323242, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8375613838434219, "clip_ratio/low_mean": 0.03689903859049082, "clip_ratio/low_min": 0.03689903859049082, "clip_ratio/high_mean": 0.09406424686312675, "clip_ratio/high_max": 0.09406424686312675, "clip_ratio/region_mean": 0.13096328545361757, "reward_total_mean": 0.7528398633003235, "reward_meter_mean": 0.8918511271476746, "reward_meter_std": 0.2632952928543091, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8825685977935791, "reward_repeat_soft_std": 0.09118378907442093, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7528398633003235, "reward_total_composite_std": 0.1126292496919632} {"timestamp_utc": "2026-04-13T01:32:43Z", "mode": "eval", "global_step": 1300, "epoch": 0.13058764439979909, "eval_loss": NaN, "eval_runtime": 43.1486, "eval_samples_per_second": 1.854, "eval_steps_per_second": 0.232, "eval_num_tokens": 2330119.0, "eval_completions/mean_length": 70.8, "eval_completions/min_length": 31.1, "eval_completions/max_length": 120.3, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 70.8, "eval_completions/min_terminated_length": 31.1, "eval_completions/max_terminated_length": 120.3, "eval_rewards/meter/mean": 0.8447652220726013, "eval_rewards/meter/std": 0.22849437817931176, "eval_rewards/count_adherence/mean": 0.9616666674613953, "eval_rewards/count_adherence/std": 0.10399272926151752, "eval_rewards/hard_gate/mean": 0.95, "eval_rewards/hard_gate/std": 0.1414213538169861, "eval_rewards/repeat_soft/mean": 0.9107989251613617, "eval_rewards/repeat_soft/std": 0.0728808406740427, "eval_rewards/judge_quality/mean": 0.40512498915195466, "eval_rewards/judge_quality/std": 0.09422349138185382, "eval_rewards/total_composite/mean": 0.7068884253501893, "eval_rewards/total_composite/std": 0.18840339481830598, "eval_reward": 0.7068884253501893, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.0746964830905199, "eval_sampling/sampling_logp_difference/max": 1.0701300621032714, "eval_sampling/importance_sampling_ratio/min": 0.34632162004709244, "eval_sampling/importance_sampling_ratio/mean": 1.021285390853882, "eval_sampling/importance_sampling_ratio/max": 1.4791025280952455, "eval_entropy": 0.8508530795574188, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7068884253501893, "eval_reward_meter_mean": 0.8447652220726013, "eval_reward_meter_std": 0.22849437817931176, "eval_reward_count_adherence_mean": 0.9616666674613953, "eval_reward_count_adherence_std": 0.10399272926151752, "eval_reward_hard_gate_mean": 0.95, "eval_reward_hard_gate_std": 0.1414213538169861, "eval_reward_repeat_soft_mean": 0.9107989251613617, "eval_reward_repeat_soft_std": 0.0728808406740427, "eval_reward_judge_quality_mean": 0.40512498915195466, "eval_reward_judge_quality_std": 0.09422349138185382, "eval_reward_total_composite_mean": 0.7068884253501893, "eval_reward_total_composite_std": 0.18840339481830598} {"timestamp_utc": "2026-04-13T01:32:53Z", "mode": "train", "global_step": 1301, "epoch": 0.1306880964339528, "loss": 0.0262, "grad_norm": 8.988228797912598, "learning_rate": 6.060606060606061e-06, "num_tokens": 2332275.0, "completions/mean_length": 84.5, "completions/min_length": 78.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.5, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.7305530309677124, "rewards/meter/std": 0.31980985403060913, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9461381435394287, "rewards/repeat_soft/std": 0.028171680867671967, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.6802377104759216, "rewards/total_composite/std": 0.1331474781036377, "reward": 0.6802377104759216, "reward_std": 0.1331474632024765, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14767470955848694, "sampling/sampling_logp_difference/max": 1.846029281616211, "sampling/importance_sampling_ratio/min": 0.15786275267601013, "sampling/importance_sampling_ratio/mean": 1.0191789865493774, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9513464346528053, "clip_ratio/low_mean": 0.04382183961570263, "clip_ratio/low_min": 0.04382183961570263, "clip_ratio/high_mean": 0.10527580976486206, "clip_ratio/high_max": 0.10527580976486206, "clip_ratio/region_mean": 0.1490976493805647, "reward_total_mean": 0.6802377104759216, "reward_meter_mean": 0.7305530309677124, "reward_meter_std": 0.31980985403060913, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9461381435394287, "reward_repeat_soft_std": 0.028171680867671967, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.6802377104759216, "reward_total_composite_std": 0.1331474781036377} {"timestamp_utc": "2026-04-13T01:33:00Z", "mode": "train", "global_step": 1302, "epoch": 0.13078854846810647, "loss": -0.0028, "grad_norm": 11.952539443969727, "learning_rate": 6.057575757575757e-06, "num_tokens": 2333837.0, "completions/mean_length": 40.25, "completions/min_length": 36.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.25, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.962089478969574, "rewards/meter/std": 0.03547055274248123, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9740467071533203, "rewards/repeat_soft/std": 0.02457144483923912, "rewards/judge_quality/mean": 0.42624998092651367, "rewards/judge_quality/std": 0.14647647738456726, "rewards/total_composite/mean": 0.7071828842163086, "rewards/total_composite/std": 0.2890150845050812, "reward": 0.7071828842163086, "reward_std": 0.2890150845050812, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12691372632980347, "sampling/sampling_logp_difference/max": 1.0749199390411377, "sampling/importance_sampling_ratio/min": 0.3413250744342804, "sampling/importance_sampling_ratio/mean": 1.0347217321395874, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0316887721419334, "clip_ratio/low_mean": 0.01689189113676548, "clip_ratio/low_min": 0.01689189113676548, "clip_ratio/high_mean": 0.10773948486894369, "clip_ratio/high_max": 0.10773948486894369, "clip_ratio/region_mean": 0.12463137600570917, "reward_total_mean": 0.7071828842163086, "reward_meter_mean": 0.962089478969574, "reward_meter_std": 0.03547055274248123, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9740467071533203, "reward_repeat_soft_std": 0.02457144483923912, "reward_judge_quality_mean": 0.42624998092651367, "reward_judge_quality_std": 0.14647647738456726, "reward_total_composite_mean": 0.7071828842163086, "reward_total_composite_std": 0.2890150845050812} {"timestamp_utc": "2026-04-13T01:33:12Z", "mode": "train", "global_step": 1303, "epoch": 0.13088900050226018, "loss": 0.0496, "grad_norm": 10.007811546325684, "learning_rate": 6.0545454545454555e-06, "num_tokens": 2336339.0, "completions/mean_length": 122.75, "completions/min_length": 106.0, "completions/max_length": 142.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.75, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 142.0, "rewards/meter/mean": 0.9351640343666077, "rewards/meter/std": 0.09180708974599838, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9345086216926575, "rewards/repeat_soft/std": 0.03547699749469757, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7700246572494507, "rewards/total_composite/std": 0.05317719653248787, "reward": 0.7700246572494507, "reward_std": 0.053177181631326675, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14079728722572327, "sampling/sampling_logp_difference/max": 2.130481719970703, "sampling/importance_sampling_ratio/min": 0.11878006160259247, "sampling/importance_sampling_ratio/mean": 1.013332724571228, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.040710061788559, "clip_ratio/low_mean": 0.04366080090403557, "clip_ratio/low_min": 0.04366080090403557, "clip_ratio/high_mean": 0.07742611598223448, "clip_ratio/high_max": 0.07742611598223448, "clip_ratio/region_mean": 0.12108691688627005, "reward_total_mean": 0.7700246572494507, "reward_meter_mean": 0.9351640343666077, "reward_meter_std": 0.09180708974599838, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9345086216926575, "reward_repeat_soft_std": 0.03547699749469757, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7700246572494507, "reward_total_composite_std": 0.05317719653248787} {"timestamp_utc": "2026-04-13T01:33:20Z", "mode": "train", "global_step": 1304, "epoch": 0.13098945253641386, "loss": 0.0043, "grad_norm": 7.787623882293701, "learning_rate": 6.051515151515152e-06, "num_tokens": 2338548.0, "completions/mean_length": 95.125, "completions/min_length": 91.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.125, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.9950034618377686, "rewards/meter/std": 0.001667112810537219, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8370051383972168, "rewards/repeat_soft/std": 0.0793335884809494, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8027645945549011, "rewards/total_composite/std": 0.019357306882739067, "reward": 0.8027645945549011, "reward_std": 0.019357310608029366, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10626053810119629, "sampling/sampling_logp_difference/max": 1.9686617851257324, "sampling/importance_sampling_ratio/min": 0.1396436095237732, "sampling/importance_sampling_ratio/mean": 1.023497462272644, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7993520498275757, "clip_ratio/low_mean": 0.03547572437673807, "clip_ratio/low_min": 0.03547572437673807, "clip_ratio/high_mean": 0.07004866190254688, "clip_ratio/high_max": 0.07004866190254688, "clip_ratio/region_mean": 0.10552438627928495, "reward_total_mean": 0.8027645945549011, "reward_meter_mean": 0.9950034618377686, "reward_meter_std": 0.001667112810537219, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8370051383972168, "reward_repeat_soft_std": 0.0793335884809494, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8027645945549011, "reward_total_composite_std": 0.019357306882739067} {"timestamp_utc": "2026-04-13T01:33:27Z", "mode": "train", "global_step": 1305, "epoch": 0.13108990457056754, "loss": 0.0699, "grad_norm": 8.415328025817871, "learning_rate": 6.048484848484849e-06, "num_tokens": 2340644.0, "completions/mean_length": 80.0, "completions/min_length": 74.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.0, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.9869159460067749, "rewards/meter/std": 0.00856485404074192, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.789980411529541, "rewards/repeat_soft/std": 0.11615831404924393, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.79273521900177, "rewards/total_composite/std": 0.025193151086568832, "reward": 0.79273521900177, "reward_std": 0.02519315667450428, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1186131164431572, "sampling/sampling_logp_difference/max": 1.2218213081359863, "sampling/importance_sampling_ratio/min": 0.2946929633617401, "sampling/importance_sampling_ratio/mean": 1.0040572881698608, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8024473339319229, "clip_ratio/low_mean": 0.025310134515166283, "clip_ratio/low_min": 0.025310134515166283, "clip_ratio/high_mean": 0.07748416531831026, "clip_ratio/high_max": 0.07748416531831026, "clip_ratio/region_mean": 0.10279429983347654, "reward_total_mean": 0.79273521900177, "reward_meter_mean": 0.9869159460067749, "reward_meter_std": 0.00856485404074192, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.789980411529541, "reward_repeat_soft_std": 0.11615831404924393, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.79273521900177, "reward_total_composite_std": 0.025193151086568832} {"timestamp_utc": "2026-04-13T01:33:35Z", "mode": "train", "global_step": 1306, "epoch": 0.13119035660472125, "loss": -0.0083, "grad_norm": 8.212944030761719, "learning_rate": 6.0454545454545456e-06, "num_tokens": 2343006.0, "completions/mean_length": 94.25, "completions/min_length": 86.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.25, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.9048924446105957, "rewards/meter/std": 0.24494385719299316, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8779935836791992, "rewards/repeat_soft/std": 0.07875936478376389, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7518759369850159, "rewards/total_composite/std": 0.10276782512664795, "reward": 0.7518759369850159, "reward_std": 0.10276784002780914, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13738125562667847, "sampling/sampling_logp_difference/max": 1.726593017578125, "sampling/importance_sampling_ratio/min": 0.17788945138454437, "sampling/importance_sampling_ratio/mean": 1.0273771286010742, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0799539312720299, "clip_ratio/low_mean": 0.0415637344121933, "clip_ratio/low_min": 0.0415637344121933, "clip_ratio/high_mean": 0.109758285805583, "clip_ratio/high_max": 0.109758285805583, "clip_ratio/region_mean": 0.1513220202177763, "reward_total_mean": 0.7518759369850159, "reward_meter_mean": 0.9048924446105957, "reward_meter_std": 0.24494385719299316, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8779935836791992, "reward_repeat_soft_std": 0.07875936478376389, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7518759369850159, "reward_total_composite_std": 0.10276782512664795} {"timestamp_utc": "2026-04-13T01:33:42Z", "mode": "train", "global_step": 1307, "epoch": 0.13129080863887493, "loss": 0.0594, "grad_norm": 13.401987075805664, "learning_rate": 6.042424242424243e-06, "num_tokens": 2344576.0, "completions/mean_length": 41.25, "completions/min_length": 31.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.25, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.9440380930900574, "rewards/meter/std": 0.07982393354177475, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9577181339263916, "rewards/repeat_soft/std": 0.034395575523376465, "rewards/judge_quality/mean": 0.5900000333786011, "rewards/judge_quality/std": 0.22696760296821594, "rewards/total_composite/mean": 0.8475889563560486, "rewards/total_composite/std": 0.056622590869665146, "reward": 0.8475889563560486, "reward_std": 0.056622590869665146, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15010479092597961, "sampling/sampling_logp_difference/max": 5.151987075805664, "sampling/importance_sampling_ratio/min": 0.0057878922671079636, "sampling/importance_sampling_ratio/mean": 1.010635256767273, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9049298018217087, "clip_ratio/low_mean": 0.10569285321980715, "clip_ratio/low_min": 0.10569285321980715, "clip_ratio/high_mean": 0.03466898947954178, "clip_ratio/high_max": 0.03466898947954178, "clip_ratio/region_mean": 0.14036184269934893, "reward_total_mean": 0.8475889563560486, "reward_meter_mean": 0.9440380930900574, "reward_meter_std": 0.07982393354177475, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9577181339263916, "reward_repeat_soft_std": 0.034395575523376465, "reward_judge_quality_mean": 0.5900000333786011, "reward_judge_quality_std": 0.22696760296821594, "reward_total_composite_mean": 0.8475889563560486, "reward_total_composite_std": 0.056622590869665146} {"timestamp_utc": "2026-04-13T01:33:49Z", "mode": "train", "global_step": 1308, "epoch": 0.13139126067302864, "loss": -0.0026, "grad_norm": 8.906615257263184, "learning_rate": 6.039393939393939e-06, "num_tokens": 2346274.0, "completions/mean_length": 65.25, "completions/min_length": 59.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.25, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9793071746826172, "rewards/meter/std": 0.02092609368264675, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9310191869735718, "rewards/repeat_soft/std": 0.035261016339063644, "rewards/judge_quality/mean": 0.5112500190734863, "rewards/judge_quality/std": 0.18216457962989807, "rewards/total_composite/mean": 0.8371651768684387, "rewards/total_composite/std": 0.06009194627404213, "reward": 0.8371651768684387, "reward_std": 0.06009193882346153, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.128183051943779, "sampling/sampling_logp_difference/max": 1.2881040573120117, "sampling/importance_sampling_ratio/min": 0.275793194770813, "sampling/importance_sampling_ratio/mean": 1.0050305128097534, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9217536300420761, "clip_ratio/low_mean": 0.07727260235697031, "clip_ratio/low_min": 0.07727260235697031, "clip_ratio/high_mean": 0.05060343537479639, "clip_ratio/high_max": 0.05060343537479639, "clip_ratio/region_mean": 0.1278760377317667, "reward_total_mean": 0.8371651768684387, "reward_meter_mean": 0.9793071746826172, "reward_meter_std": 0.02092609368264675, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9310191869735718, "reward_repeat_soft_std": 0.035261016339063644, "reward_judge_quality_mean": 0.5112500190734863, "reward_judge_quality_std": 0.18216457962989807, "reward_total_composite_mean": 0.8371651768684387, "reward_total_composite_std": 0.06009194627404213} {"timestamp_utc": "2026-04-13T01:33:56Z", "mode": "train", "global_step": 1309, "epoch": 0.13149171270718232, "loss": 0.0417, "grad_norm": 14.117969512939453, "learning_rate": 6.0363636363636365e-06, "num_tokens": 2347901.0, "completions/mean_length": 41.375, "completions/min_length": 33.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.375, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.8665153980255127, "rewards/meter/std": 0.2836450934410095, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9580709338188171, "rewards/repeat_soft/std": 0.05692196637392044, "rewards/judge_quality/mean": 0.36374998092651367, "rewards/judge_quality/std": 0.09500939399003983, "rewards/total_composite/mean": 0.7448639869689941, "rewards/total_composite/std": 0.1410859227180481, "reward": 0.7448639869689941, "reward_std": 0.1410859227180481, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13370774686336517, "sampling/sampling_logp_difference/max": 2.1684517860412598, "sampling/importance_sampling_ratio/min": 0.11435452848672867, "sampling/importance_sampling_ratio/mean": 1.0069283246994019, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0122607424855232, "clip_ratio/low_mean": 0.019886363297700882, "clip_ratio/low_min": 0.019886363297700882, "clip_ratio/high_mean": 0.14590988401323557, "clip_ratio/high_max": 0.14590988401323557, "clip_ratio/region_mean": 0.16579624731093645, "reward_total_mean": 0.7448639869689941, "reward_meter_mean": 0.8665153980255127, "reward_meter_std": 0.2836450934410095, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9580709338188171, "reward_repeat_soft_std": 0.05692196637392044, "reward_judge_quality_mean": 0.36374998092651367, "reward_judge_quality_std": 0.09500939399003983, "reward_total_composite_mean": 0.7448639869689941, "reward_total_composite_std": 0.1410859227180481} {"timestamp_utc": "2026-04-13T01:34:04Z", "mode": "train", "global_step": 1310, "epoch": 0.131592164741336, "loss": 0.0297, "grad_norm": 12.77451229095459, "learning_rate": 6.033333333333335e-06, "num_tokens": 2349611.0, "completions/mean_length": 46.75, "completions/min_length": 44.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.75, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.6915186643600464, "rewards/meter/std": 0.3996758460998535, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9210962653160095, "rewards/repeat_soft/std": 0.04986434057354927, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.6804180145263672, "rewards/total_composite/std": 0.1776176393032074, "reward": 0.6804180145263672, "reward_std": 0.1776176393032074, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08138161897659302, "sampling/sampling_logp_difference/max": 1.159752607345581, "sampling/importance_sampling_ratio/min": 0.3135637640953064, "sampling/importance_sampling_ratio/mean": 1.0042967796325684, "sampling/importance_sampling_ratio/max": 1.9930415153503418, "entropy": 0.6367932558059692, "clip_ratio/low_mean": 0.038807546719908714, "clip_ratio/low_min": 0.038807546719908714, "clip_ratio/high_mean": 0.02707886602729559, "clip_ratio/high_max": 0.02707886602729559, "clip_ratio/region_mean": 0.0658864127472043, "reward_total_mean": 0.6804180145263672, "reward_meter_mean": 0.6915186643600464, "reward_meter_std": 0.3996758460998535, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9210962653160095, "reward_repeat_soft_std": 0.04986434057354927, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.6804180145263672, "reward_total_composite_std": 0.1776176393032074} {"timestamp_utc": "2026-04-13T01:34:11Z", "mode": "train", "global_step": 1311, "epoch": 0.1316926167754897, "loss": 0.0301, "grad_norm": 9.820587158203125, "learning_rate": 6.030303030303031e-06, "num_tokens": 2351768.0, "completions/mean_length": 63.625, "completions/min_length": 58.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.625, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.782690167427063, "rewards/meter/std": 0.15685687959194183, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8870646357536316, "rewards/repeat_soft/std": 0.044491373002529144, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.773167073726654, "rewards/total_composite/std": 0.05264436826109886, "reward": 0.773167073726654, "reward_std": 0.05264437571167946, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08810056746006012, "sampling/sampling_logp_difference/max": 1.3431243896484375, "sampling/importance_sampling_ratio/min": 0.26102882623672485, "sampling/importance_sampling_ratio/mean": 1.0164246559143066, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.493072047829628, "clip_ratio/low_mean": 0.017045455053448677, "clip_ratio/low_min": 0.017045455053448677, "clip_ratio/high_mean": 0.06370748579502106, "clip_ratio/high_max": 0.06370748579502106, "clip_ratio/region_mean": 0.08075294084846973, "reward_total_mean": 0.773167073726654, "reward_meter_mean": 0.782690167427063, "reward_meter_std": 0.15685687959194183, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8870646357536316, "reward_repeat_soft_std": 0.044491373002529144, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.773167073726654, "reward_total_composite_std": 0.05264436826109886} {"timestamp_utc": "2026-04-13T01:34:18Z", "mode": "train", "global_step": 1312, "epoch": 0.1317930688096434, "loss": 0.0067, "grad_norm": 10.206457138061523, "learning_rate": 6.027272727272728e-06, "num_tokens": 2353469.0, "completions/mean_length": 40.625, "completions/min_length": 38.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.625, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.9843893051147461, "rewards/meter/std": 0.004676532931625843, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9559552073478699, "rewards/repeat_soft/std": 0.040307603776454926, "rewards/judge_quality/mean": 0.42124998569488525, "rewards/judge_quality/std": 0.0699872374534607, "rewards/total_composite/mean": 0.8149456977844238, "rewards/total_composite/std": 0.021126186475157738, "reward": 0.8149456977844238, "reward_std": 0.021126171573996544, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11584191769361496, "sampling/sampling_logp_difference/max": 2.514019727706909, "sampling/importance_sampling_ratio/min": 0.08094222098588943, "sampling/importance_sampling_ratio/mean": 1.0237773656845093, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8275373578071594, "clip_ratio/low_mean": 0.033799534663558006, "clip_ratio/low_min": 0.033799534663558006, "clip_ratio/high_mean": 0.10226181708276272, "clip_ratio/high_max": 0.10226181708276272, "clip_ratio/region_mean": 0.13606135174632072, "reward_total_mean": 0.8149456977844238, "reward_meter_mean": 0.9843893051147461, "reward_meter_std": 0.004676532931625843, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9559552073478699, "reward_repeat_soft_std": 0.040307603776454926, "reward_judge_quality_mean": 0.42124998569488525, "reward_judge_quality_std": 0.0699872374534607, "reward_total_composite_mean": 0.8149456977844238, "reward_total_composite_std": 0.021126186475157738} {"timestamp_utc": "2026-04-13T01:34:24Z", "mode": "train", "global_step": 1313, "epoch": 0.13189352084379707, "loss": 0.0611, "grad_norm": 13.047164916992188, "learning_rate": 6.024242424242425e-06, "num_tokens": 2355100.0, "completions/mean_length": 31.875, "completions/min_length": 29.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.875, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.8282817602157593, "rewards/meter/std": 0.17143860459327698, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9188117980957031, "rewards/repeat_soft/std": 0.07968474179506302, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.7781079411506653, "rewards/total_composite/std": 0.11878776550292969, "reward": 0.7781079411506653, "reward_std": 0.1187877506017685, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11081690341234207, "sampling/sampling_logp_difference/max": 1.4982519149780273, "sampling/importance_sampling_ratio/min": 0.22352056205272675, "sampling/importance_sampling_ratio/mean": 0.9965724945068359, "sampling/importance_sampling_ratio/max": 1.883732557296753, "entropy": 0.5949073396623135, "clip_ratio/low_mean": 0.03766989801079035, "clip_ratio/low_min": 0.03766989801079035, "clip_ratio/high_mean": 0.08015585225075483, "clip_ratio/high_max": 0.08015585225075483, "clip_ratio/region_mean": 0.11782575026154518, "reward_total_mean": 0.7781079411506653, "reward_meter_mean": 0.8282817602157593, "reward_meter_std": 0.17143860459327698, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9188117980957031, "reward_repeat_soft_std": 0.07968474179506302, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.7781079411506653, "reward_total_composite_std": 0.11878776550292969} {"timestamp_utc": "2026-04-13T01:34:31Z", "mode": "train", "global_step": 1314, "epoch": 0.13199397287795078, "loss": 0.0384, "grad_norm": 13.060811042785645, "learning_rate": 6.021212121212122e-06, "num_tokens": 2356712.0, "completions/mean_length": 44.5, "completions/min_length": 43.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.5, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9886382818222046, "rewards/meter/std": 0.014332191087305546, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9823265075683594, "rewards/repeat_soft/std": 0.016347724944353104, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.12631450593471527, "rewards/total_composite/mean": 0.7924948930740356, "rewards/total_composite/std": 0.03828875720500946, "reward": 0.7924948930740356, "reward_std": 0.03828875720500946, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10968618839979172, "sampling/sampling_logp_difference/max": 1.5331544876098633, "sampling/importance_sampling_ratio/min": 0.21585369110107422, "sampling/importance_sampling_ratio/mean": 1.0163145065307617, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7508808076381683, "clip_ratio/low_mean": 0.03386698244139552, "clip_ratio/low_min": 0.03386698244139552, "clip_ratio/high_mean": 0.056989927776157856, "clip_ratio/high_max": 0.056989927776157856, "clip_ratio/region_mean": 0.09085691021755338, "reward_total_mean": 0.7924948930740356, "reward_meter_mean": 0.9886382818222046, "reward_meter_std": 0.014332191087305546, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9823265075683594, "reward_repeat_soft_std": 0.016347724944353104, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.12631450593471527, "reward_total_composite_mean": 0.7924948930740356, "reward_total_composite_std": 0.03828875720500946} {"timestamp_utc": "2026-04-13T01:34:38Z", "mode": "train", "global_step": 1315, "epoch": 0.13209442491210446, "loss": 0.0835, "grad_norm": 22.21614646911621, "learning_rate": 6.018181818181818e-06, "num_tokens": 2358134.0, "completions/mean_length": 29.75, "completions/min_length": 25.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.75, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.7531710267066956, "rewards/meter/std": 0.3706286549568176, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9695032835006714, "rewards/repeat_soft/std": 0.042702946811914444, "rewards/judge_quality/mean": 0.8312500715255737, "rewards/judge_quality/std": 0.09538456052541733, "rewards/total_composite/mean": 0.8352522850036621, "rewards/total_composite/std": 0.1702217161655426, "reward": 0.8352522850036621, "reward_std": 0.1702217012643814, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2078274041414261, "sampling/sampling_logp_difference/max": 1.5511493682861328, "sampling/importance_sampling_ratio/min": 0.2120041698217392, "sampling/importance_sampling_ratio/mean": 0.9993975162506104, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0394762381911278, "clip_ratio/low_mean": 0.033350841142237186, "clip_ratio/low_min": 0.033350841142237186, "clip_ratio/high_mean": 0.11543934047222137, "clip_ratio/high_max": 0.11543934047222137, "clip_ratio/region_mean": 0.14879018161445856, "reward_total_mean": 0.8352522850036621, "reward_meter_mean": 0.7531710267066956, "reward_meter_std": 0.3706286549568176, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9695032835006714, "reward_repeat_soft_std": 0.042702946811914444, "reward_judge_quality_mean": 0.8312500715255737, "reward_judge_quality_std": 0.09538456052541733, "reward_total_composite_mean": 0.8352522850036621, "reward_total_composite_std": 0.1702217161655426} {"timestamp_utc": "2026-04-13T01:34:46Z", "mode": "train", "global_step": 1316, "epoch": 0.13219487694625817, "loss": 0.0238, "grad_norm": 8.857390403747559, "learning_rate": 6.015151515151516e-06, "num_tokens": 2360073.0, "completions/mean_length": 75.375, "completions/min_length": 65.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.375, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.919614315032959, "rewards/meter/std": 0.11661911010742188, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8322687149047852, "rewards/repeat_soft/std": 0.1191723644733429, "rewards/judge_quality/mean": 0.35999998450279236, "rewards/judge_quality/std": 0.09165150672197342, "rewards/total_composite/mean": 0.7550532817840576, "rewards/total_composite/std": 0.05366427078843117, "reward": 0.7550532817840576, "reward_std": 0.05366426706314087, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13664498925209045, "sampling/sampling_logp_difference/max": 1.2015981674194336, "sampling/importance_sampling_ratio/min": 0.3007132411003113, "sampling/importance_sampling_ratio/mean": 1.0254497528076172, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2666477262973785, "clip_ratio/low_mean": 0.028566016349941492, "clip_ratio/low_min": 0.028566016349941492, "clip_ratio/high_mean": 0.0662228548899293, "clip_ratio/high_max": 0.0662228548899293, "clip_ratio/region_mean": 0.09478887123987079, "reward_total_mean": 0.7550532817840576, "reward_meter_mean": 0.919614315032959, "reward_meter_std": 0.11661911010742188, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8322687149047852, "reward_repeat_soft_std": 0.1191723644733429, "reward_judge_quality_mean": 0.35999998450279236, "reward_judge_quality_std": 0.09165150672197342, "reward_total_composite_mean": 0.7550532817840576, "reward_total_composite_std": 0.05366427078843117} {"timestamp_utc": "2026-04-13T01:34:54Z", "mode": "train", "global_step": 1317, "epoch": 0.13229532898041185, "loss": 0.022, "grad_norm": 11.40386962890625, "learning_rate": 6.012121212121213e-06, "num_tokens": 2362158.0, "completions/mean_length": 87.625, "completions/min_length": 77.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.625, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.8023767471313477, "rewards/meter/std": 0.2849498987197876, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9635120630264282, "rewards/repeat_soft/std": 0.01309928484261036, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.733420729637146, "rewards/total_composite/std": 0.12805728614330292, "reward": 0.733420729637146, "reward_std": 0.12805728614330292, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.131356880068779, "sampling/sampling_logp_difference/max": 1.8352714776992798, "sampling/importance_sampling_ratio/min": 0.15957017242908478, "sampling/importance_sampling_ratio/mean": 1.008623719215393, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7654286324977875, "clip_ratio/low_mean": 0.04955300223082304, "clip_ratio/low_min": 0.04955300223082304, "clip_ratio/high_mean": 0.06846249382942915, "clip_ratio/high_max": 0.06846249382942915, "clip_ratio/region_mean": 0.11801549606025219, "reward_total_mean": 0.733420729637146, "reward_meter_mean": 0.8023767471313477, "reward_meter_std": 0.2849498987197876, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9635120630264282, "reward_repeat_soft_std": 0.01309928484261036, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.733420729637146, "reward_total_composite_std": 0.12805728614330292} {"timestamp_utc": "2026-04-13T01:35:06Z", "mode": "train", "global_step": 1318, "epoch": 0.13239578101456553, "loss": -0.1027, "grad_norm": 2.367274522781372, "learning_rate": 6.00909090909091e-06, "num_tokens": 2363673.0, "completions/mean_length": 173.375, "completions/min_length": 55.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 60.5, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.8370566368103027, "rewards/meter/std": 0.2520429193973541, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.8902388215065002, "rewards/repeat_soft/std": 0.09601560235023499, "rewards/judge_quality/mean": 0.16250000894069672, "rewards/judge_quality/std": 0.10264363139867783, "rewards/total_composite/mean": 0.4217807650566101, "rewards/total_composite/std": 0.3694986402988434, "reward": 0.4217807650566101, "reward_std": 0.3694986402988434, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13581886887550354, "sampling/sampling_logp_difference/max": 1.702265739440918, "sampling/importance_sampling_ratio/min": 0.1822700798511505, "sampling/importance_sampling_ratio/mean": 1.0166465044021606, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7851816043257713, "clip_ratio/low_mean": 0.022408293560147285, "clip_ratio/low_min": 0.022408293560147285, "clip_ratio/high_mean": 0.0712656769901514, "clip_ratio/high_max": 0.0712656769901514, "clip_ratio/region_mean": 0.09367397055029869, "reward_total_mean": 0.4217807650566101, "reward_meter_mean": 0.8370566368103027, "reward_meter_std": 0.2520429193973541, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.8902388215065002, "reward_repeat_soft_std": 0.09601560235023499, "reward_judge_quality_mean": 0.16250000894069672, "reward_judge_quality_std": 0.10264363139867783, "reward_total_composite_mean": 0.4217807650566101, "reward_total_composite_std": 0.3694986402988434} {"timestamp_utc": "2026-04-13T01:35:13Z", "mode": "train", "global_step": 1319, "epoch": 0.13249623304871924, "loss": -0.059, "grad_norm": 9.05211067199707, "learning_rate": 6.0060606060606065e-06, "num_tokens": 2365662.0, "completions/mean_length": 74.625, "completions/min_length": 64.0, "completions/max_length": 89.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.625, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 89.0, "rewards/meter/mean": 0.9648871421813965, "rewards/meter/std": 0.08086027950048447, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9538623690605164, "rewards/repeat_soft/std": 0.030417844653129578, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7710854411125183, "rewards/total_composite/std": 0.04575014114379883, "reward": 0.7710854411125183, "reward_std": 0.04575014114379883, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1157589852809906, "sampling/sampling_logp_difference/max": 1.5472097396850586, "sampling/importance_sampling_ratio/min": 0.21284101903438568, "sampling/importance_sampling_ratio/mean": 1.0340903997421265, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8586238697171211, "clip_ratio/low_mean": 0.031455863267183304, "clip_ratio/low_min": 0.031455863267183304, "clip_ratio/high_mean": 0.08085490996018052, "clip_ratio/high_max": 0.08085490996018052, "clip_ratio/region_mean": 0.11231077322736382, "reward_total_mean": 0.7710854411125183, "reward_meter_mean": 0.9648871421813965, "reward_meter_std": 0.08086027950048447, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9538623690605164, "reward_repeat_soft_std": 0.030417844653129578, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7710854411125183, "reward_total_composite_std": 0.04575014114379883} {"timestamp_utc": "2026-04-13T01:35:20Z", "mode": "train", "global_step": 1320, "epoch": 0.13259668508287292, "loss": -0.0215, "grad_norm": 17.555906295776367, "learning_rate": 6.003030303030304e-06, "num_tokens": 2367088.0, "completions/mean_length": 31.25, "completions/min_length": 29.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.25, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.28938397765159607, "rewards/meter/std": 0.3103305399417877, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9528172016143799, "rewards/repeat_soft/std": 0.03941875323653221, "rewards/judge_quality/mean": 0.59375, "rewards/judge_quality/std": 0.22965426743030548, "rewards/total_composite/mean": 0.5536295175552368, "rewards/total_composite/std": 0.11205455660820007, "reward": 0.5536295175552368, "reward_std": 0.11205456405878067, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11481940001249313, "sampling/sampling_logp_difference/max": 1.0912046432495117, "sampling/importance_sampling_ratio/min": 0.33581170439720154, "sampling/importance_sampling_ratio/mean": 0.9847961664199829, "sampling/importance_sampling_ratio/max": 1.7894421815872192, "entropy": 0.5554206669330597, "clip_ratio/low_mean": 0.05357668688520789, "clip_ratio/low_min": 0.05357668688520789, "clip_ratio/high_mean": 0.06282467627897859, "clip_ratio/high_max": 0.06282467627897859, "clip_ratio/region_mean": 0.11640136316418648, "reward_total_mean": 0.5536295175552368, "reward_meter_mean": 0.28938397765159607, "reward_meter_std": 0.3103305399417877, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9528172016143799, "reward_repeat_soft_std": 0.03941875323653221, "reward_judge_quality_mean": 0.59375, "reward_judge_quality_std": 0.22965426743030548, "reward_total_composite_mean": 0.5536295175552368, "reward_total_composite_std": 0.11205455660820007} {"timestamp_utc": "2026-04-13T01:35:27Z", "mode": "train", "global_step": 1321, "epoch": 0.13269713711702663, "loss": 0.2112, "grad_norm": 26.548049926757812, "learning_rate": 6e-06, "num_tokens": 2368665.0, "completions/mean_length": 21.125, "completions/min_length": 15.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.125, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.6046105027198792, "rewards/meter/std": 0.4359692931175232, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9559118747711182, "rewards/repeat_soft/std": 0.009517517872154713, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.6481659412384033, "rewards/total_composite/std": 0.21512356400489807, "reward": 0.6481659412384033, "reward_std": 0.21512354910373688, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15486420691013336, "sampling/sampling_logp_difference/max": 0.9168505668640137, "sampling/importance_sampling_ratio/min": 0.3997761309146881, "sampling/importance_sampling_ratio/mean": 1.0286279916763306, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1631434485316277, "clip_ratio/low_mean": 0.07599206548184156, "clip_ratio/low_min": 0.07599206548184156, "clip_ratio/high_mean": 0.09409348014742136, "clip_ratio/high_max": 0.09409348014742136, "clip_ratio/region_mean": 0.17008554562926292, "reward_total_mean": 0.6481659412384033, "reward_meter_mean": 0.6046105027198792, "reward_meter_std": 0.4359692931175232, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9559118747711182, "reward_repeat_soft_std": 0.009517517872154713, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.6481659412384033, "reward_total_composite_std": 0.21512356400489807} {"timestamp_utc": "2026-04-13T01:35:39Z", "mode": "train", "global_step": 1322, "epoch": 0.1327975891511803, "loss": 0.0321, "grad_norm": 5.444184303283691, "learning_rate": 5.996969696969697e-06, "num_tokens": 2370916.0, "completions/mean_length": 149.375, "completions/min_length": 93.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 97.5714340209961, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.8688409924507141, "rewards/meter/std": 0.28579387068748474, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9209449291229248, "rewards/repeat_soft/std": 0.05123668164014816, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.18845234811306, "rewards/total_composite/mean": 0.7500729560852051, "rewards/total_composite/std": 0.14054282009601593, "reward": 0.7500729560852051, "reward_std": 0.14054280519485474, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14674323797225952, "sampling/sampling_logp_difference/max": 1.8925886154174805, "sampling/importance_sampling_ratio/min": 0.1506812572479248, "sampling/importance_sampling_ratio/mean": 1.018070936203003, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.973444975912571, "clip_ratio/low_mean": 0.03684394992887974, "clip_ratio/low_min": 0.03684394992887974, "clip_ratio/high_mean": 0.07348132226616144, "clip_ratio/high_max": 0.07348132226616144, "clip_ratio/region_mean": 0.11032527219504118, "reward_total_mean": 0.7500729560852051, "reward_meter_mean": 0.8688409924507141, "reward_meter_std": 0.28579387068748474, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9209449291229248, "reward_repeat_soft_std": 0.05123668164014816, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.18845234811306, "reward_total_composite_mean": 0.7500729560852051, "reward_total_composite_std": 0.14054282009601593} {"timestamp_utc": "2026-04-13T01:35:46Z", "mode": "train", "global_step": 1323, "epoch": 0.132898041185334, "loss": 0.0615, "grad_norm": 13.80717945098877, "learning_rate": 5.993939393939394e-06, "num_tokens": 2372430.0, "completions/mean_length": 46.25, "completions/min_length": 36.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.25, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.8281381130218506, "rewards/meter/std": 0.3427148163318634, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9390705823898315, "rewards/repeat_soft/std": 0.06682360917329788, "rewards/judge_quality/mean": 0.47749999165534973, "rewards/judge_quality/std": 0.16960039734840393, "rewards/total_composite/mean": 0.7410692572593689, "rewards/total_composite/std": 0.13441656529903412, "reward": 0.7410692572593689, "reward_std": 0.1344165802001953, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14878657460212708, "sampling/sampling_logp_difference/max": 1.2680778503417969, "sampling/importance_sampling_ratio/min": 0.28137195110321045, "sampling/importance_sampling_ratio/mean": 1.0207276344299316, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9955775439739227, "clip_ratio/low_mean": 0.06410006852820516, "clip_ratio/low_min": 0.06410006852820516, "clip_ratio/high_mean": 0.0721121272072196, "clip_ratio/high_max": 0.0721121272072196, "clip_ratio/region_mean": 0.13621219573542476, "reward_total_mean": 0.7410692572593689, "reward_meter_mean": 0.8281381130218506, "reward_meter_std": 0.3427148163318634, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9390705823898315, "reward_repeat_soft_std": 0.06682360917329788, "reward_judge_quality_mean": 0.47749999165534973, "reward_judge_quality_std": 0.16960039734840393, "reward_total_composite_mean": 0.7410692572593689, "reward_total_composite_std": 0.13441656529903412} {"timestamp_utc": "2026-04-13T01:35:53Z", "mode": "train", "global_step": 1324, "epoch": 0.1329984932194877, "loss": 0.1159, "grad_norm": 14.034634590148926, "learning_rate": 5.990909090909092e-06, "num_tokens": 2374115.0, "completions/mean_length": 43.625, "completions/min_length": 37.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.625, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.7026336193084717, "rewards/meter/std": 0.41234877705574036, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9629354476928711, "rewards/repeat_soft/std": 0.032081738114356995, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.7271037101745605, "rewards/total_composite/std": 0.2228107899427414, "reward": 0.7271037101745605, "reward_std": 0.2228107899427414, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1634635180234909, "sampling/sampling_logp_difference/max": 2.383573293685913, "sampling/importance_sampling_ratio/min": 0.09222045540809631, "sampling/importance_sampling_ratio/mean": 1.0025755167007446, "sampling/importance_sampling_ratio/max": 1.9986613988876343, "entropy": 1.1801003590226173, "clip_ratio/low_mean": 0.04848471935838461, "clip_ratio/low_min": 0.04848471935838461, "clip_ratio/high_mean": 0.08714693505316973, "clip_ratio/high_max": 0.08714693505316973, "clip_ratio/region_mean": 0.13563165441155434, "reward_total_mean": 0.7271037101745605, "reward_meter_mean": 0.7026336193084717, "reward_meter_std": 0.41234877705574036, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9629354476928711, "reward_repeat_soft_std": 0.032081738114356995, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.7271037101745605, "reward_total_composite_std": 0.2228107899427414} {"timestamp_utc": "2026-04-13T01:36:05Z", "mode": "train", "global_step": 1325, "epoch": 0.13309894525364138, "loss": -0.1295, "grad_norm": 2.1374800205230713, "learning_rate": 5.987878787878788e-06, "num_tokens": 2376071.0, "completions/mean_length": 128.5, "completions/min_length": 44.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 73.71428680419922, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.8674899339675903, "rewards/meter/std": 0.23637078702449799, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7749499082565308, "rewards/repeat_soft/std": 0.1403287649154663, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.22952981293201447, "rewards/total_composite/mean": 0.7356154918670654, "rewards/total_composite/std": 0.1674242913722992, "reward": 0.7356154918670654, "reward_std": 0.1674242913722992, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12533241510391235, "sampling/sampling_logp_difference/max": 2.919178009033203, "sampling/importance_sampling_ratio/min": 0.05397803708910942, "sampling/importance_sampling_ratio/mean": 1.0018547773361206, "sampling/importance_sampling_ratio/max": 1.8582799434661865, "entropy": 0.7982849068939686, "clip_ratio/low_mean": 0.024738955311477184, "clip_ratio/low_min": 0.024738955311477184, "clip_ratio/high_mean": 0.06751331500709057, "clip_ratio/high_max": 0.06751331500709057, "clip_ratio/region_mean": 0.09225227031856775, "reward_total_mean": 0.7356154918670654, "reward_meter_mean": 0.8674899339675903, "reward_meter_std": 0.23637078702449799, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7749499082565308, "reward_repeat_soft_std": 0.1403287649154663, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.22952981293201447, "reward_total_composite_mean": 0.7356154918670654, "reward_total_composite_std": 0.1674242913722992} {"timestamp_utc": "2026-04-13T01:36:17Z", "mode": "train", "global_step": 1326, "epoch": 0.1331993972877951, "loss": -0.1024, "grad_norm": 2.9685006141662598, "learning_rate": 5.984848484848486e-06, "num_tokens": 2377916.0, "completions/mean_length": 101.625, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 43.000003814697266, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.7461591958999634, "rewards/meter/std": 0.40999314188957214, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9750214219093323, "rewards/repeat_soft/std": 0.02348564751446247, "rewards/judge_quality/mean": 0.3512499928474426, "rewards/judge_quality/std": 0.15797263383865356, "rewards/total_composite/mean": 0.6524457931518555, "rewards/total_composite/std": 0.31119945645332336, "reward": 0.6524457931518555, "reward_std": 0.311199426651001, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16490411758422852, "sampling/sampling_logp_difference/max": 2.152737617492676, "sampling/importance_sampling_ratio/min": 0.11616570502519608, "sampling/importance_sampling_ratio/mean": 1.0151557922363281, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.013551041483879, "clip_ratio/low_mean": 0.00872093066573143, "clip_ratio/low_min": 0.00872093066573143, "clip_ratio/high_mean": 0.13466470036655664, "clip_ratio/high_max": 0.13466470036655664, "clip_ratio/region_mean": 0.14338563103228807, "reward_total_mean": 0.6524457931518555, "reward_meter_mean": 0.7461591958999634, "reward_meter_std": 0.40999314188957214, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9750214219093323, "reward_repeat_soft_std": 0.02348564751446247, "reward_judge_quality_mean": 0.3512499928474426, "reward_judge_quality_std": 0.15797263383865356, "reward_total_composite_mean": 0.6524457931518555, "reward_total_composite_std": 0.31119945645332336} {"timestamp_utc": "2026-04-13T01:36:28Z", "mode": "train", "global_step": 1327, "epoch": 0.13329984932194877, "loss": -0.0586, "grad_norm": 2.7939090728759766, "learning_rate": 5.981818181818182e-06, "num_tokens": 2379233.0, "completions/mean_length": 80.625, "completions/min_length": 17.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 19.0, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.7096605896949768, "rewards/meter/std": 0.3933052718639374, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9485969543457031, "rewards/repeat_soft/std": 0.021013345569372177, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.6373006105422974, "rewards/total_composite/std": 0.2835118770599365, "reward": 0.6373006105422974, "reward_std": 0.28351184725761414, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15236569941043854, "sampling/sampling_logp_difference/max": 1.248920202255249, "sampling/importance_sampling_ratio/min": 0.2868143320083618, "sampling/importance_sampling_ratio/mean": 1.0355474948883057, "sampling/importance_sampling_ratio/max": 1.7207502126693726, "entropy": 0.9741610288619995, "clip_ratio/low_mean": 0.04861111240461469, "clip_ratio/low_min": 0.04861111240461469, "clip_ratio/high_mean": 0.08605843735858798, "clip_ratio/high_max": 0.08605843735858798, "clip_ratio/region_mean": 0.13466954976320267, "reward_total_mean": 0.6373006105422974, "reward_meter_mean": 0.7096605896949768, "reward_meter_std": 0.3933052718639374, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9485969543457031, "reward_repeat_soft_std": 0.021013345569372177, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.6373006105422974, "reward_total_composite_std": 0.2835118770599365} {"timestamp_utc": "2026-04-13T01:36:40Z", "mode": "train", "global_step": 1328, "epoch": 0.13340030135610245, "loss": -0.0931, "grad_norm": 2.438215732574463, "learning_rate": 5.978787878787879e-06, "num_tokens": 2380769.0, "completions/mean_length": 160.0, "completions/min_length": 36.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 42.66666793823242, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.5438164472579956, "rewards/meter/std": 0.4780561625957489, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9787144660949707, "rewards/repeat_soft/std": 0.022013681009411812, "rewards/judge_quality/mean": 0.3725000023841858, "rewards/judge_quality/std": 0.22282280027866364, "rewards/total_composite/mean": 0.5337594747543335, "rewards/total_composite/std": 0.3733977973461151, "reward": 0.5337594747543335, "reward_std": 0.3733978271484375, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15256337821483612, "sampling/sampling_logp_difference/max": 1.7879364490509033, "sampling/importance_sampling_ratio/min": 0.16730506718158722, "sampling/importance_sampling_ratio/mean": 1.007826566696167, "sampling/importance_sampling_ratio/max": 1.9419182538986206, "entropy": 0.8008382543921471, "clip_ratio/low_mean": 0.021839595399796963, "clip_ratio/low_min": 0.021839595399796963, "clip_ratio/high_mean": 0.06393287237733603, "clip_ratio/high_max": 0.06393287237733603, "clip_ratio/region_mean": 0.08577246777713299, "reward_total_mean": 0.5337594747543335, "reward_meter_mean": 0.5438164472579956, "reward_meter_std": 0.4780561625957489, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9787144660949707, "reward_repeat_soft_std": 0.022013681009411812, "reward_judge_quality_mean": 0.3725000023841858, "reward_judge_quality_std": 0.22282280027866364, "reward_total_composite_mean": 0.5337594747543335, "reward_total_composite_std": 0.3733977973461151} {"timestamp_utc": "2026-04-13T01:36:47Z", "mode": "train", "global_step": 1329, "epoch": 0.13350075339025616, "loss": -0.0291, "grad_norm": 18.613998413085938, "learning_rate": 5.975757575757576e-06, "num_tokens": 2382042.0, "completions/mean_length": 21.125, "completions/min_length": 18.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.125, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.9591783881187439, "rewards/meter/std": 0.09055937081575394, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9593343734741211, "rewards/repeat_soft/std": 0.008953613229095936, "rewards/judge_quality/mean": 0.39249998331069946, "rewards/judge_quality/std": 0.08892211318016052, "rewards/total_composite/mean": 0.7953137159347534, "rewards/total_composite/std": 0.045516110956668854, "reward": 0.7953137159347534, "reward_std": 0.045516107231378555, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15630343556404114, "sampling/sampling_logp_difference/max": 1.644382357597351, "sampling/importance_sampling_ratio/min": 0.19313180446624756, "sampling/importance_sampling_ratio/mean": 1.0149158239364624, "sampling/importance_sampling_ratio/max": 1.6566463708877563, "entropy": 1.2428646758198738, "clip_ratio/low_mean": 0.0496031753718853, "clip_ratio/low_min": 0.0496031753718853, "clip_ratio/high_mean": 0.07528235856443644, "clip_ratio/high_max": 0.07528235856443644, "clip_ratio/region_mean": 0.12488553393632174, "reward_total_mean": 0.7953137159347534, "reward_meter_mean": 0.9591783881187439, "reward_meter_std": 0.09055937081575394, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9593343734741211, "reward_repeat_soft_std": 0.008953613229095936, "reward_judge_quality_mean": 0.39249998331069946, "reward_judge_quality_std": 0.08892211318016052, "reward_total_composite_mean": 0.7953137159347534, "reward_total_composite_std": 0.045516110956668854} {"timestamp_utc": "2026-04-13T01:36:54Z", "mode": "train", "global_step": 1330, "epoch": 0.13360120542440984, "loss": -0.0133, "grad_norm": 16.632043838500977, "learning_rate": 5.972727272727274e-06, "num_tokens": 2383371.0, "completions/mean_length": 20.125, "completions/min_length": 16.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.125, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.9828482866287231, "rewards/meter/std": 0.020903030410408974, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9482327103614807, "rewards/repeat_soft/std": 0.02057761140167713, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8198549747467041, "rewards/total_composite/std": 0.00884835422039032, "reward": 0.8198549747467041, "reward_std": 0.00884835422039032, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12784725427627563, "sampling/sampling_logp_difference/max": 1.5882482528686523, "sampling/importance_sampling_ratio/min": 0.20428313314914703, "sampling/importance_sampling_ratio/mean": 1.0153214931488037, "sampling/importance_sampling_ratio/max": 1.6563072204589844, "entropy": 0.780411384999752, "clip_ratio/low_mean": 0.038815789856016636, "clip_ratio/low_min": 0.038815789856016636, "clip_ratio/high_mean": 0.09232143126428127, "clip_ratio/high_max": 0.09232143126428127, "clip_ratio/region_mean": 0.1311372211202979, "reward_total_mean": 0.8198549747467041, "reward_meter_mean": 0.9828482866287231, "reward_meter_std": 0.020903030410408974, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9482327103614807, "reward_repeat_soft_std": 0.02057761140167713, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8198549747467041, "reward_total_composite_std": 0.00884835422039032} {"timestamp_utc": "2026-04-13T01:37:01Z", "mode": "train", "global_step": 1331, "epoch": 0.13370165745856355, "loss": 0.0223, "grad_norm": 14.397272109985352, "learning_rate": 5.96969696969697e-06, "num_tokens": 2385285.0, "completions/mean_length": 83.25, "completions/min_length": 77.0, "completions/max_length": 89.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.25, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 89.0, "rewards/meter/mean": 0.6761950254440308, "rewards/meter/std": 0.2844979763031006, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9599876999855042, "rewards/repeat_soft/std": 0.030876046046614647, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.676286518573761, "rewards/total_composite/std": 0.12685246765613556, "reward": 0.676286518573761, "reward_std": 0.12685245275497437, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12160026282072067, "sampling/sampling_logp_difference/max": 1.5839390754699707, "sampling/importance_sampling_ratio/min": 0.2051653414964676, "sampling/importance_sampling_ratio/mean": 1.0065182447433472, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7250372692942619, "clip_ratio/low_mean": 0.036764475516974926, "clip_ratio/low_min": 0.036764475516974926, "clip_ratio/high_mean": 0.08375677280128002, "clip_ratio/high_max": 0.08375677280128002, "clip_ratio/region_mean": 0.12052124831825495, "reward_total_mean": 0.676286518573761, "reward_meter_mean": 0.6761950254440308, "reward_meter_std": 0.2844979763031006, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9599876999855042, "reward_repeat_soft_std": 0.030876046046614647, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.676286518573761, "reward_total_composite_std": 0.12685246765613556} {"timestamp_utc": "2026-04-13T01:37:09Z", "mode": "train", "global_step": 1332, "epoch": 0.13380210949271723, "loss": 0.0245, "grad_norm": 18.09085464477539, "learning_rate": 5.966666666666667e-06, "num_tokens": 2386947.0, "completions/mean_length": 31.75, "completions/min_length": 25.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.75, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.5837363004684448, "rewards/meter/std": 0.44545501470565796, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9504238963127136, "rewards/repeat_soft/std": 0.055679693818092346, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.13866712152957916, "rewards/total_composite/mean": 0.6397237181663513, "rewards/total_composite/std": 0.22100239992141724, "reward": 0.6397237181663513, "reward_std": 0.22100239992141724, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17629414796829224, "sampling/sampling_logp_difference/max": 2.08819317817688, "sampling/importance_sampling_ratio/min": 0.1239108219742775, "sampling/importance_sampling_ratio/mean": 0.988405704498291, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9978543408215046, "clip_ratio/low_mean": 0.045831285417079926, "clip_ratio/low_min": 0.045831285417079926, "clip_ratio/high_mean": 0.14947726763784885, "clip_ratio/high_max": 0.14947726763784885, "clip_ratio/region_mean": 0.19530855305492878, "reward_total_mean": 0.6397237181663513, "reward_meter_mean": 0.5837363004684448, "reward_meter_std": 0.44545501470565796, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9504238963127136, "reward_repeat_soft_std": 0.055679693818092346, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.13866712152957916, "reward_total_composite_mean": 0.6397237181663513, "reward_total_composite_std": 0.22100239992141724} {"timestamp_utc": "2026-04-13T01:37:21Z", "mode": "train", "global_step": 1333, "epoch": 0.1339025615268709, "loss": -0.1777, "grad_norm": 2.1128480434417725, "learning_rate": 5.963636363636364e-06, "num_tokens": 2389140.0, "completions/mean_length": 145.125, "completions/min_length": 76.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 92.71428680419922, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.8164526224136353, "rewards/meter/std": 0.2375941276550293, "rewards/count_adherence/mean": 0.8500000238418579, "rewards/count_adherence/std": 0.20701967179775238, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8207929134368896, "rewards/repeat_soft/std": 0.10648838430643082, "rewards/judge_quality/mean": 0.24500000476837158, "rewards/judge_quality/std": 0.15537971258163452, "rewards/total_composite/mean": 0.5976240038871765, "rewards/total_composite/std": 0.2519613802433014, "reward": 0.5976240038871765, "reward_std": 0.2519613802433014, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10425487160682678, "sampling/sampling_logp_difference/max": 1.8712677955627441, "sampling/importance_sampling_ratio/min": 0.15392839908599854, "sampling/importance_sampling_ratio/mean": 1.0081874132156372, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5321333557367325, "clip_ratio/low_mean": 0.009868420660495758, "clip_ratio/low_min": 0.009868420660495758, "clip_ratio/high_mean": 0.06813804479315877, "clip_ratio/high_max": 0.06813804479315877, "clip_ratio/region_mean": 0.07800646545365453, "reward_total_mean": 0.5976240038871765, "reward_meter_mean": 0.8164526224136353, "reward_meter_std": 0.2375941276550293, "reward_count_adherence_mean": 0.8500000238418579, "reward_count_adherence_std": 0.20701967179775238, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8207929134368896, "reward_repeat_soft_std": 0.10648838430643082, "reward_judge_quality_mean": 0.24500000476837158, "reward_judge_quality_std": 0.15537971258163452, "reward_total_composite_mean": 0.5976240038871765, "reward_total_composite_std": 0.2519613802433014} {"timestamp_utc": "2026-04-13T01:37:34Z", "mode": "train", "global_step": 1334, "epoch": 0.13400301356102462, "loss": -0.1864, "grad_norm": 2.57804274559021, "learning_rate": 5.960606060606061e-06, "num_tokens": 2391364.0, "completions/mean_length": 149.0, "completions/min_length": 88.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 97.14286041259766, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.7674049139022827, "rewards/meter/std": 0.26274555921554565, "rewards/count_adherence/mean": 0.8958333134651184, "rewards/count_adherence/std": 0.294627845287323, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9246237874031067, "rewards/repeat_soft/std": 0.04615677893161774, "rewards/judge_quality/mean": 0.4112499952316284, "rewards/judge_quality/std": 0.1797965168952942, "rewards/total_composite/mean": 0.6265805959701538, "rewards/total_composite/std": 0.2807009816169739, "reward": 0.6265805959701538, "reward_std": 0.2807009816169739, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10285772383213043, "sampling/sampling_logp_difference/max": 1.47344970703125, "sampling/importance_sampling_ratio/min": 0.22913366556167603, "sampling/importance_sampling_ratio/mean": 1.0183663368225098, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6086663529276848, "clip_ratio/low_mean": 0.019886363297700882, "clip_ratio/low_min": 0.019886363297700882, "clip_ratio/high_mean": 0.07078679371625185, "clip_ratio/high_max": 0.07078679371625185, "clip_ratio/region_mean": 0.09067315701395273, "reward_total_mean": 0.6265805959701538, "reward_meter_mean": 0.7674049139022827, "reward_meter_std": 0.26274555921554565, "reward_count_adherence_mean": 0.8958333134651184, "reward_count_adherence_std": 0.294627845287323, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9246237874031067, "reward_repeat_soft_std": 0.04615677893161774, "reward_judge_quality_mean": 0.4112499952316284, "reward_judge_quality_std": 0.1797965168952942, "reward_total_composite_mean": 0.6265805959701538, "reward_total_composite_std": 0.2807009816169739} {"timestamp_utc": "2026-04-13T01:37:45Z", "mode": "train", "global_step": 1335, "epoch": 0.1341034655951783, "loss": -0.1581, "grad_norm": 2.877767562866211, "learning_rate": 5.9575757575757575e-06, "num_tokens": 2393502.0, "completions/mean_length": 138.25, "completions/min_length": 67.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 84.85714721679688, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.5181673765182495, "rewards/meter/std": 0.3302626609802246, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.2828427255153656, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9180291891098022, "rewards/repeat_soft/std": 0.054461587220430374, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.3147788345813751, "rewards/total_composite/mean": 0.574171245098114, "rewards/total_composite/std": 0.24504175782203674, "reward": 0.574171245098114, "reward_std": 0.24504175782203674, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1473846286535263, "sampling/sampling_logp_difference/max": 1.6147518157958984, "sampling/importance_sampling_ratio/min": 0.19894003868103027, "sampling/importance_sampling_ratio/mean": 1.01116943359375, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7391086742281914, "clip_ratio/low_mean": 0.0150602413341403, "clip_ratio/low_min": 0.0150602413341403, "clip_ratio/high_mean": 0.12408160138875246, "clip_ratio/high_max": 0.12408160138875246, "clip_ratio/region_mean": 0.13914184272289276, "reward_total_mean": 0.574171245098114, "reward_meter_mean": 0.5181673765182495, "reward_meter_std": 0.3302626609802246, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.2828427255153656, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9180291891098022, "reward_repeat_soft_std": 0.054461587220430374, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.3147788345813751, "reward_total_composite_mean": 0.574171245098114, "reward_total_composite_std": 0.24504175782203674} {"timestamp_utc": "2026-04-13T01:37:53Z", "mode": "train", "global_step": 1336, "epoch": 0.13420391762933198, "loss": 0.0056, "grad_norm": 9.907699584960938, "learning_rate": 5.954545454545455e-06, "num_tokens": 2395669.0, "completions/mean_length": 86.875, "completions/min_length": 80.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.875, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.9769899249076843, "rewards/meter/std": 0.03450459986925125, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9294705390930176, "rewards/repeat_soft/std": 0.05124903470277786, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7958425283432007, "rewards/total_composite/std": 0.02886941283941269, "reward": 0.7958425283432007, "reward_std": 0.02886941097676754, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1238371804356575, "sampling/sampling_logp_difference/max": 1.958765983581543, "sampling/importance_sampling_ratio/min": 0.1410323530435562, "sampling/importance_sampling_ratio/mean": 1.015339970588684, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9123608246445656, "clip_ratio/low_mean": 0.03712925501167774, "clip_ratio/low_min": 0.03712925501167774, "clip_ratio/high_mean": 0.09349368885159492, "clip_ratio/high_max": 0.09349368885159492, "clip_ratio/region_mean": 0.13062294386327267, "reward_total_mean": 0.7958425283432007, "reward_meter_mean": 0.9769899249076843, "reward_meter_std": 0.03450459986925125, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9294705390930176, "reward_repeat_soft_std": 0.05124903470277786, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7958425283432007, "reward_total_composite_std": 0.02886941283941269} {"timestamp_utc": "2026-04-13T01:38:05Z", "mode": "train", "global_step": 1337, "epoch": 0.1343043696634857, "loss": -0.0812, "grad_norm": 1.6962616443634033, "learning_rate": 5.951515151515151e-06, "num_tokens": 2397320.0, "completions/mean_length": 158.375, "completions/min_length": 33.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 40.5, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.730595588684082, "rewards/meter/std": 0.38234829902648926, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.930479884147644, "rewards/repeat_soft/std": 0.06648889183998108, "rewards/judge_quality/mean": 0.2887499928474426, "rewards/judge_quality/std": 0.166770800948143, "rewards/total_composite/mean": 0.6257348656654358, "rewards/total_composite/std": 0.29077619314193726, "reward": 0.6257348656654358, "reward_std": 0.29077616333961487, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17596745491027832, "sampling/sampling_logp_difference/max": 5.017612457275391, "sampling/importance_sampling_ratio/min": 0.006620313972234726, "sampling/importance_sampling_ratio/mean": 1.0097776651382446, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.837487667798996, "clip_ratio/low_mean": 0.01944444514811039, "clip_ratio/low_min": 0.01944444514811039, "clip_ratio/high_mean": 0.0765467188321054, "clip_ratio/high_max": 0.0765467188321054, "clip_ratio/region_mean": 0.09599116398021579, "reward_total_mean": 0.6257348656654358, "reward_meter_mean": 0.730595588684082, "reward_meter_std": 0.38234829902648926, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.930479884147644, "reward_repeat_soft_std": 0.06648889183998108, "reward_judge_quality_mean": 0.2887499928474426, "reward_judge_quality_std": 0.166770800948143, "reward_total_composite_mean": 0.6257348656654358, "reward_total_composite_std": 0.29077619314193726} {"timestamp_utc": "2026-04-13T01:38:11Z", "mode": "train", "global_step": 1338, "epoch": 0.13440482169763937, "loss": 0.0171, "grad_norm": 11.198810577392578, "learning_rate": 5.948484848484849e-06, "num_tokens": 2398884.0, "completions/mean_length": 41.5, "completions/min_length": 39.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.5, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.9849881529808044, "rewards/meter/std": 0.005699469242244959, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9414850473403931, "rewards/repeat_soft/std": 0.0379074290394783, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.25150617957115173, "rewards/total_composite/mean": 0.8445181846618652, "rewards/total_composite/std": 0.07740126550197601, "reward": 0.8445181846618652, "reward_std": 0.07740127295255661, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10386534035205841, "sampling/sampling_logp_difference/max": 1.4646692276000977, "sampling/importance_sampling_ratio/min": 0.2311544418334961, "sampling/importance_sampling_ratio/mean": 1.0114495754241943, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7391662895679474, "clip_ratio/low_mean": 0.041106811026111245, "clip_ratio/low_min": 0.041106811026111245, "clip_ratio/high_mean": 0.030230187810957432, "clip_ratio/high_max": 0.030230187810957432, "clip_ratio/region_mean": 0.07133699883706868, "reward_total_mean": 0.8445181846618652, "reward_meter_mean": 0.9849881529808044, "reward_meter_std": 0.005699469242244959, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9414850473403931, "reward_repeat_soft_std": 0.0379074290394783, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.25150617957115173, "reward_total_composite_mean": 0.8445181846618652, "reward_total_composite_std": 0.07740126550197601} {"timestamp_utc": "2026-04-13T01:38:18Z", "mode": "train", "global_step": 1339, "epoch": 0.13450527373179308, "loss": 0.0996, "grad_norm": 14.866206169128418, "learning_rate": 5.9454545454545465e-06, "num_tokens": 2400423.0, "completions/mean_length": 39.375, "completions/min_length": 34.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.375, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.8629153966903687, "rewards/meter/std": 0.21370987594127655, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8228611946105957, "rewards/repeat_soft/std": 0.08787237852811813, "rewards/judge_quality/mean": 0.39375001192092896, "rewards/judge_quality/std": 0.09941794723272324, "rewards/total_composite/mean": 0.7387230396270752, "rewards/total_composite/std": 0.09309365600347519, "reward": 0.7387230396270752, "reward_std": 0.0930936336517334, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11183873564004898, "sampling/sampling_logp_difference/max": 2.177255153656006, "sampling/importance_sampling_ratio/min": 0.1133522316813469, "sampling/importance_sampling_ratio/mean": 1.0127313137054443, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7160054035484791, "clip_ratio/low_mean": 0.03830151446163654, "clip_ratio/low_min": 0.03830151446163654, "clip_ratio/high_mean": 0.07074965070933104, "clip_ratio/high_max": 0.07074965070933104, "clip_ratio/region_mean": 0.10905116517096758, "reward_total_mean": 0.7387230396270752, "reward_meter_mean": 0.8629153966903687, "reward_meter_std": 0.21370987594127655, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8228611946105957, "reward_repeat_soft_std": 0.08787237852811813, "reward_judge_quality_mean": 0.39375001192092896, "reward_judge_quality_std": 0.09941794723272324, "reward_total_composite_mean": 0.7387230396270752, "reward_total_composite_std": 0.09309365600347519} {"timestamp_utc": "2026-04-13T01:38:29Z", "mode": "train", "global_step": 1340, "epoch": 0.13460572576594676, "loss": -0.2288, "grad_norm": 2.1284022331237793, "learning_rate": 5.942424242424243e-06, "num_tokens": 2403091.0, "completions/mean_length": 189.5, "completions/min_length": 134.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 143.42857360839844, "completions/min_terminated_length": 134.0, "completions/max_terminated_length": 166.0, "rewards/meter/mean": 0.915316104888916, "rewards/meter/std": 0.14162999391555786, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8150745630264282, "rewards/repeat_soft/std": 0.07602003216743469, "rewards/judge_quality/mean": 0.2724999785423279, "rewards/judge_quality/std": 0.16104568541049957, "rewards/total_composite/mean": 0.6624072790145874, "rewards/total_composite/std": 0.27033108472824097, "reward": 0.6624072790145874, "reward_std": 0.2703310549259186, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10947053134441376, "sampling/sampling_logp_difference/max": 1.7451972961425781, "sampling/importance_sampling_ratio/min": 0.17461052536964417, "sampling/importance_sampling_ratio/mean": 1.0102870464324951, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.769673116505146, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08685469441115856, "clip_ratio/high_max": 0.08685469441115856, "clip_ratio/region_mean": 0.08685469441115856, "reward_total_mean": 0.6624072790145874, "reward_meter_mean": 0.915316104888916, "reward_meter_std": 0.14162999391555786, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8150745630264282, "reward_repeat_soft_std": 0.07602003216743469, "reward_judge_quality_mean": 0.2724999785423279, "reward_judge_quality_std": 0.16104568541049957, "reward_total_composite_mean": 0.6624072790145874, "reward_total_composite_std": 0.27033108472824097} {"timestamp_utc": "2026-04-13T01:38:36Z", "mode": "train", "global_step": 1341, "epoch": 0.13470617780010044, "loss": -0.032, "grad_norm": 17.530099868774414, "learning_rate": 5.93939393939394e-06, "num_tokens": 2404513.0, "completions/mean_length": 20.75, "completions/min_length": 17.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.75, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.9892822504043579, "rewards/meter/std": 0.008297543972730637, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9495933055877686, "rewards/repeat_soft/std": 0.02156112529337406, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.8393863439559937, "rewards/total_composite/std": 0.0508783683180809, "reward": 0.8393863439559937, "reward_std": 0.05087835714221001, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13932763040065765, "sampling/sampling_logp_difference/max": 1.256777286529541, "sampling/importance_sampling_ratio/min": 0.2845696210861206, "sampling/importance_sampling_ratio/mean": 1.0225541591644287, "sampling/importance_sampling_ratio/max": 1.6340998411178589, "entropy": 0.7894485108554363, "clip_ratio/low_mean": 0.12365377135574818, "clip_ratio/low_min": 0.12365377135574818, "clip_ratio/high_mean": 0.0052083334885537624, "clip_ratio/high_max": 0.0052083334885537624, "clip_ratio/region_mean": 0.12886210484430194, "reward_total_mean": 0.8393863439559937, "reward_meter_mean": 0.9892822504043579, "reward_meter_std": 0.008297543972730637, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9495933055877686, "reward_repeat_soft_std": 0.02156112529337406, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.8393863439559937, "reward_total_composite_std": 0.0508783683180809} {"timestamp_utc": "2026-04-13T01:38:44Z", "mode": "train", "global_step": 1342, "epoch": 0.13480662983425415, "loss": 0.0582, "grad_norm": 11.32034969329834, "learning_rate": 5.936363636363637e-06, "num_tokens": 2406274.0, "completions/mean_length": 53.125, "completions/min_length": 43.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.125, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.8474869728088379, "rewards/meter/std": 0.23476096987724304, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.927552342414856, "rewards/repeat_soft/std": 0.0305857565253973, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7336243391036987, "rewards/total_composite/std": 0.09783148020505905, "reward": 0.7336243391036987, "reward_std": 0.09783148020505905, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12665154039859772, "sampling/sampling_logp_difference/max": 3.2333450317382812, "sampling/importance_sampling_ratio/min": 0.03942539915442467, "sampling/importance_sampling_ratio/mean": 1.0241175889968872, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7870222926139832, "clip_ratio/low_mean": 0.03902821335941553, "clip_ratio/low_min": 0.03902821335941553, "clip_ratio/high_mean": 0.08704834617674351, "clip_ratio/high_max": 0.08704834617674351, "clip_ratio/region_mean": 0.12607655953615904, "reward_total_mean": 0.7336243391036987, "reward_meter_mean": 0.8474869728088379, "reward_meter_std": 0.23476096987724304, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.927552342414856, "reward_repeat_soft_std": 0.0305857565253973, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7336243391036987, "reward_total_composite_std": 0.09783148020505905} {"timestamp_utc": "2026-04-13T01:38:51Z", "mode": "train", "global_step": 1343, "epoch": 0.13490708186840783, "loss": 0.0319, "grad_norm": 10.89278793334961, "learning_rate": 5.933333333333335e-06, "num_tokens": 2407980.0, "completions/mean_length": 43.25, "completions/min_length": 41.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.25, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.9903115034103394, "rewards/meter/std": 0.007458357606083155, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9829654693603516, "rewards/repeat_soft/std": 0.024379469454288483, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.8596867322921753, "rewards/total_composite/std": 0.07056183367967606, "reward": 0.8596867322921753, "reward_std": 0.07056181877851486, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11272593587636948, "sampling/sampling_logp_difference/max": 1.789555549621582, "sampling/importance_sampling_ratio/min": 0.16703440248966217, "sampling/importance_sampling_ratio/mean": 0.9997726082801819, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6977763473987579, "clip_ratio/low_mean": 0.07714030425995588, "clip_ratio/low_min": 0.07714030425995588, "clip_ratio/high_mean": 0.031976744532585144, "clip_ratio/high_max": 0.031976744532585144, "clip_ratio/region_mean": 0.10911704879254103, "reward_total_mean": 0.8596867322921753, "reward_meter_mean": 0.9903115034103394, "reward_meter_std": 0.007458357606083155, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9829654693603516, "reward_repeat_soft_std": 0.024379469454288483, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.8596867322921753, "reward_total_composite_std": 0.07056183367967606} {"timestamp_utc": "2026-04-13T01:38:58Z", "mode": "train", "global_step": 1344, "epoch": 0.13500753390256154, "loss": 0.0369, "grad_norm": 15.463644981384277, "learning_rate": 5.93030303030303e-06, "num_tokens": 2409551.0, "completions/mean_length": 44.375, "completions/min_length": 41.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.375, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.8518151044845581, "rewards/meter/std": 0.3216133117675781, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9880026578903198, "rewards/repeat_soft/std": 0.015359737910330296, "rewards/judge_quality/mean": 0.5212500095367432, "rewards/judge_quality/std": 0.14156650006771088, "rewards/total_composite/mean": 0.7884920835494995, "rewards/total_composite/std": 0.16110320389270782, "reward": 0.7884920835494995, "reward_std": 0.16110318899154663, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1421390175819397, "sampling/sampling_logp_difference/max": 1.8886561393737793, "sampling/importance_sampling_ratio/min": 0.15127496421337128, "sampling/importance_sampling_ratio/mean": 1.001012921333313, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7908435612916946, "clip_ratio/low_mean": 0.043889736756682396, "clip_ratio/low_min": 0.043889736756682396, "clip_ratio/high_mean": 0.0934942802414298, "clip_ratio/high_max": 0.0934942802414298, "clip_ratio/region_mean": 0.1373840169981122, "reward_total_mean": 0.7884920835494995, "reward_meter_mean": 0.8518151044845581, "reward_meter_std": 0.3216133117675781, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9880026578903198, "reward_repeat_soft_std": 0.015359737910330296, "reward_judge_quality_mean": 0.5212500095367432, "reward_judge_quality_std": 0.14156650006771088, "reward_total_composite_mean": 0.7884920835494995, "reward_total_composite_std": 0.16110320389270782} {"timestamp_utc": "2026-04-13T01:39:06Z", "mode": "train", "global_step": 1345, "epoch": 0.13510798593671522, "loss": 0.1105, "grad_norm": 9.981226921081543, "learning_rate": 5.927272727272728e-06, "num_tokens": 2411461.0, "completions/mean_length": 67.75, "completions/min_length": 58.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.75, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.9694181084632874, "rewards/meter/std": 0.04306391626596451, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.859180212020874, "rewards/repeat_soft/std": 0.10198870301246643, "rewards/judge_quality/mean": 0.30124998092651367, "rewards/judge_quality/std": 0.10398317128419876, "rewards/total_composite/mean": 0.7500311136245728, "rewards/total_composite/std": 0.040714625269174576, "reward": 0.7500311136245728, "reward_std": 0.040714625269174576, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11013256013393402, "sampling/sampling_logp_difference/max": 1.2646780014038086, "sampling/importance_sampling_ratio/min": 0.282330185174942, "sampling/importance_sampling_ratio/mean": 1.017654538154602, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.917246975004673, "clip_ratio/low_mean": 0.05347340926527977, "clip_ratio/low_min": 0.05347340926527977, "clip_ratio/high_mean": 0.05821990733966231, "clip_ratio/high_max": 0.05821990733966231, "clip_ratio/region_mean": 0.11169331660494208, "reward_total_mean": 0.7500311136245728, "reward_meter_mean": 0.9694181084632874, "reward_meter_std": 0.04306391626596451, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.859180212020874, "reward_repeat_soft_std": 0.10198870301246643, "reward_judge_quality_mean": 0.30124998092651367, "reward_judge_quality_std": 0.10398317128419876, "reward_total_composite_mean": 0.7500311136245728, "reward_total_composite_std": 0.040714625269174576} {"timestamp_utc": "2026-04-13T01:39:13Z", "mode": "train", "global_step": 1346, "epoch": 0.1352084379708689, "loss": 0.0291, "grad_norm": 7.954216003417969, "learning_rate": 5.924242424242425e-06, "num_tokens": 2413616.0, "completions/mean_length": 89.375, "completions/min_length": 78.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.375, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.7597460746765137, "rewards/meter/std": 0.3087576925754547, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.878596842288971, "rewards/repeat_soft/std": 0.024035152047872543, "rewards/judge_quality/mean": 0.45249998569488525, "rewards/judge_quality/std": 0.18100909888744354, "rewards/total_composite/mean": 0.7154954671859741, "rewards/total_composite/std": 0.16661247611045837, "reward": 0.7154954671859741, "reward_std": 0.16661246120929718, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1027073860168457, "sampling/sampling_logp_difference/max": 1.32915461063385, "sampling/importance_sampling_ratio/min": 0.2647009491920471, "sampling/importance_sampling_ratio/mean": 1.013420820236206, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7814216390252113, "clip_ratio/low_mean": 0.039329035207629204, "clip_ratio/low_min": 0.039329035207629204, "clip_ratio/high_mean": 0.07595761027187109, "clip_ratio/high_max": 0.07595761027187109, "clip_ratio/region_mean": 0.1152866454795003, "reward_total_mean": 0.7154954671859741, "reward_meter_mean": 0.7597460746765137, "reward_meter_std": 0.3087576925754547, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.878596842288971, "reward_repeat_soft_std": 0.024035152047872543, "reward_judge_quality_mean": 0.45249998569488525, "reward_judge_quality_std": 0.18100909888744354, "reward_total_composite_mean": 0.7154954671859741, "reward_total_composite_std": 0.16661247611045837} {"timestamp_utc": "2026-04-13T01:39:20Z", "mode": "train", "global_step": 1347, "epoch": 0.1353088900050226, "loss": 0.0546, "grad_norm": 14.287421226501465, "learning_rate": 5.921212121212122e-06, "num_tokens": 2415339.0, "completions/mean_length": 43.375, "completions/min_length": 40.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.375, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.9811661243438721, "rewards/meter/std": 0.02622855082154274, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.972983181476593, "rewards/repeat_soft/std": 0.026861680671572685, "rewards/judge_quality/mean": 0.375, "rewards/judge_quality/std": 0.1035098284482956, "rewards/total_composite/mean": 0.8013230562210083, "rewards/total_composite/std": 0.030430208891630173, "reward": 0.8013230562210083, "reward_std": 0.030430205166339874, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16539105772972107, "sampling/sampling_logp_difference/max": 1.868232250213623, "sampling/importance_sampling_ratio/min": 0.15439635515213013, "sampling/importance_sampling_ratio/mean": 1.0128551721572876, "sampling/importance_sampling_ratio/max": 1.8078595399856567, "entropy": 1.1070107892155647, "clip_ratio/low_mean": 0.05362395476549864, "clip_ratio/low_min": 0.05362395476549864, "clip_ratio/high_mean": 0.06861723680049181, "clip_ratio/high_max": 0.06861723680049181, "clip_ratio/region_mean": 0.12224119156599045, "reward_total_mean": 0.8013230562210083, "reward_meter_mean": 0.9811661243438721, "reward_meter_std": 0.02622855082154274, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.972983181476593, "reward_repeat_soft_std": 0.026861680671572685, "reward_judge_quality_mean": 0.375, "reward_judge_quality_std": 0.1035098284482956, "reward_total_composite_mean": 0.8013230562210083, "reward_total_composite_std": 0.030430208891630173} {"timestamp_utc": "2026-04-13T01:39:26Z", "mode": "train", "global_step": 1348, "epoch": 0.1354093420391763, "loss": 0.0145, "grad_norm": 22.779136657714844, "learning_rate": 5.9181818181818184e-06, "num_tokens": 2416693.0, "completions/mean_length": 19.25, "completions/min_length": 16.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.25, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.756381630897522, "rewards/meter/std": 0.34163299202919006, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9359691739082336, "rewards/repeat_soft/std": 0.030012985691428185, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.697218656539917, "rewards/total_composite/std": 0.1451389640569687, "reward": 0.697218656539917, "reward_std": 0.1451389640569687, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12292852997779846, "sampling/sampling_logp_difference/max": 0.7122793197631836, "sampling/importance_sampling_ratio/min": 0.49052485823631287, "sampling/importance_sampling_ratio/mean": 1.0364038944244385, "sampling/importance_sampling_ratio/max": 1.9375132322311401, "entropy": 1.035483606159687, "clip_ratio/low_mean": 0.03267045505344868, "clip_ratio/low_min": 0.03267045505344868, "clip_ratio/high_mean": 0.09684659168124199, "clip_ratio/high_max": 0.09684659168124199, "clip_ratio/region_mean": 0.12951704673469067, "reward_total_mean": 0.697218656539917, "reward_meter_mean": 0.756381630897522, "reward_meter_std": 0.34163299202919006, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9359691739082336, "reward_repeat_soft_std": 0.030012985691428185, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.697218656539917, "reward_total_composite_std": 0.1451389640569687} {"timestamp_utc": "2026-04-13T01:39:38Z", "mode": "train", "global_step": 1349, "epoch": 0.13550979407333, "loss": 0.1902, "grad_norm": 4.3265604972839355, "learning_rate": 5.915151515151516e-06, "num_tokens": 2419281.0, "completions/mean_length": 172.5, "completions/min_length": 67.0, "completions/max_length": 505.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 172.5, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 505.0, "rewards/meter/mean": 0.7526119351387024, "rewards/meter/std": 0.33061501383781433, "rewards/count_adherence/mean": 0.3125, "rewards/count_adherence/std": 0.25877460837364197, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7846164107322693, "rewards/repeat_soft/std": 0.18280591070652008, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.5798870325088501, "rewards/total_composite/std": 0.14645648002624512, "reward": 0.5798870325088501, "reward_std": 0.14645648002624512, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08532381057739258, "sampling/sampling_logp_difference/max": 1.37331223487854, "sampling/importance_sampling_ratio/min": 0.25326669216156006, "sampling/importance_sampling_ratio/mean": 1.0097967386245728, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7576210685074329, "clip_ratio/low_mean": 0.029832308180630207, "clip_ratio/low_min": 0.029832308180630207, "clip_ratio/high_mean": 0.04311298532411456, "clip_ratio/high_max": 0.04311298532411456, "clip_ratio/region_mean": 0.07294529350474477, "reward_total_mean": 0.5798870325088501, "reward_meter_mean": 0.7526119351387024, "reward_meter_std": 0.33061501383781433, "reward_count_adherence_mean": 0.3125, "reward_count_adherence_std": 0.25877460837364197, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7846164107322693, "reward_repeat_soft_std": 0.18280591070652008, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.5798870325088501, "reward_total_composite_std": 0.14645648002624512} {"timestamp_utc": "2026-04-13T01:39:50Z", "mode": "train", "global_step": 1350, "epoch": 0.13561024610748368, "loss": -0.0923, "grad_norm": 1.5356287956237793, "learning_rate": 5.912121212121212e-06, "num_tokens": 2420846.0, "completions/mean_length": 156.625, "completions/min_length": 31.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 38.16666793823242, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.6478617191314697, "rewards/meter/std": 0.3822675347328186, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.815968930721283, "rewards/repeat_soft/std": 0.07774503529071808, "rewards/judge_quality/mean": 0.26999998092651367, "rewards/judge_quality/std": 0.17968225479125977, "rewards/total_composite/mean": 0.49431946873664856, "rewards/total_composite/std": 0.3353841006755829, "reward": 0.49431946873664856, "reward_std": 0.3353841006755829, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11179220676422119, "sampling/sampling_logp_difference/max": 3.348447799682617, "sampling/importance_sampling_ratio/min": 0.03513885661959648, "sampling/importance_sampling_ratio/mean": 1.0450634956359863, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6309277266263962, "clip_ratio/low_mean": 0.02187499962747097, "clip_ratio/low_min": 0.02187499962747097, "clip_ratio/high_mean": 0.04170275363139808, "clip_ratio/high_max": 0.04170275363139808, "clip_ratio/region_mean": 0.06357775325886905, "reward_total_mean": 0.49431946873664856, "reward_meter_mean": 0.6478617191314697, "reward_meter_std": 0.3822675347328186, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.815968930721283, "reward_repeat_soft_std": 0.07774503529071808, "reward_judge_quality_mean": 0.26999998092651367, "reward_judge_quality_std": 0.17968225479125977, "reward_total_composite_mean": 0.49431946873664856, "reward_total_composite_std": 0.3353841006755829} {"timestamp_utc": "2026-04-13T01:41:00Z", "mode": "eval", "global_step": 1350, "epoch": 0.13561024610748368, "eval_loss": NaN, "eval_runtime": 70.4058, "eval_samples_per_second": 1.136, "eval_steps_per_second": 0.142, "eval_num_tokens": 2420846.0, "eval_completions/mean_length": 112.3375, "eval_completions/min_length": 30.2, "eval_completions/max_length": 325.7, "eval_completions/clipped_ratio": 0.0875, "eval_completions/mean_terminated_length": 73.42357292175294, "eval_completions/min_terminated_length": 30.2, "eval_completions/max_terminated_length": 132.4, "eval_rewards/meter/mean": 0.8243018686771393, "eval_rewards/meter/std": 0.2504229828715324, "eval_rewards/count_adherence/mean": 0.9302083313465118, "eval_rewards/count_adherence/std": 0.16290227472782134, "eval_rewards/hard_gate/mean": 0.9125, "eval_rewards/hard_gate/std": 0.19317627549171448, "eval_rewards/repeat_soft/mean": 0.8512749969959259, "eval_rewards/repeat_soft/std": 0.1319960318505764, "eval_rewards/judge_quality/mean": 0.35974999964237214, "eval_rewards/judge_quality/std": 0.14537042006850243, "eval_rewards/total_composite/mean": 0.6766853898763656, "eval_rewards/total_composite/std": 0.203667027130723, "eval_reward": 0.6766853898763656, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.07026319019496441, "eval_sampling/sampling_logp_difference/max": 1.1197218418121337, "eval_sampling/importance_sampling_ratio/min": 0.3347224235534668, "eval_sampling/importance_sampling_ratio/mean": 1.0167441725730897, "eval_sampling/importance_sampling_ratio/max": 1.5128594040870667, "eval_entropy": 0.7724981606006622, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6766853898763656, "eval_reward_meter_mean": 0.8243018686771393, "eval_reward_meter_std": 0.2504229828715324, "eval_reward_count_adherence_mean": 0.9302083313465118, "eval_reward_count_adherence_std": 0.16290227472782134, "eval_reward_hard_gate_mean": 0.9125, "eval_reward_hard_gate_std": 0.19317627549171448, "eval_reward_repeat_soft_mean": 0.8512749969959259, "eval_reward_repeat_soft_std": 0.1319960318505764, "eval_reward_judge_quality_mean": 0.35974999964237214, "eval_reward_judge_quality_std": 0.14537042006850243, "eval_reward_total_composite_mean": 0.6766853898763656, "eval_reward_total_composite_std": 0.203667027130723} {"timestamp_utc": "2026-04-13T01:41:10Z", "mode": "train", "global_step": 1351, "epoch": 0.13571069814163736, "loss": 0.0388, "grad_norm": 22.321392059326172, "learning_rate": 5.90909090909091e-06, "num_tokens": 2422206.0, "completions/mean_length": 21.0, "completions/min_length": 19.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.0, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.9913536310195923, "rewards/meter/std": 0.005532666575163603, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8845176696777344, "rewards/repeat_soft/std": 0.08242236077785492, "rewards/judge_quality/mean": 0.2837499976158142, "rewards/judge_quality/std": 0.136165589094162, "rewards/total_composite/mean": 0.7696858644485474, "rewards/total_composite/std": 0.041346482932567596, "reward": 0.7696858644485474, "reward_std": 0.041346464306116104, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0905856117606163, "sampling/sampling_logp_difference/max": 0.8700160980224609, "sampling/importance_sampling_ratio/min": 0.4189448058605194, "sampling/importance_sampling_ratio/mean": 1.015699863433838, "sampling/importance_sampling_ratio/max": 1.6428371667861938, "entropy": 0.7505421862006187, "clip_ratio/low_mean": 0.05438460409641266, "clip_ratio/low_min": 0.05438460409641266, "clip_ratio/high_mean": 0.041566986590623856, "clip_ratio/high_max": 0.041566986590623856, "clip_ratio/region_mean": 0.09595159068703651, "reward_total_mean": 0.7696858644485474, "reward_meter_mean": 0.9913536310195923, "reward_meter_std": 0.005532666575163603, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8845176696777344, "reward_repeat_soft_std": 0.08242236077785492, "reward_judge_quality_mean": 0.2837499976158142, "reward_judge_quality_std": 0.136165589094162, "reward_total_composite_mean": 0.7696858644485474, "reward_total_composite_std": 0.041346482932567596} {"timestamp_utc": "2026-04-13T01:41:17Z", "mode": "train", "global_step": 1352, "epoch": 0.13581115017579107, "loss": 0.0057, "grad_norm": 12.882585525512695, "learning_rate": 5.906060606060607e-06, "num_tokens": 2423876.0, "completions/mean_length": 44.75, "completions/min_length": 39.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.75, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.9096282124519348, "rewards/meter/std": 0.23132354021072388, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8977269530296326, "rewards/repeat_soft/std": 0.05445731058716774, "rewards/judge_quality/mean": 0.5900000333786011, "rewards/judge_quality/std": 0.27994900941848755, "rewards/total_composite/mean": 0.8261053562164307, "rewards/total_composite/std": 0.1501709669828415, "reward": 0.8261053562164307, "reward_std": 0.1501709669828415, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12366481125354767, "sampling/sampling_logp_difference/max": 1.902524709701538, "sampling/importance_sampling_ratio/min": 0.1491914689540863, "sampling/importance_sampling_ratio/mean": 1.0056755542755127, "sampling/importance_sampling_ratio/max": 1.962833046913147, "entropy": 0.7375760078430176, "clip_ratio/low_mean": 0.06676682597026229, "clip_ratio/low_min": 0.06676682597026229, "clip_ratio/high_mean": 0.04656815342605114, "clip_ratio/high_max": 0.04656815342605114, "clip_ratio/region_mean": 0.11333497939631343, "reward_total_mean": 0.8261053562164307, "reward_meter_mean": 0.9096282124519348, "reward_meter_std": 0.23132354021072388, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8977269530296326, "reward_repeat_soft_std": 0.05445731058716774, "reward_judge_quality_mean": 0.5900000333786011, "reward_judge_quality_std": 0.27994900941848755, "reward_total_composite_mean": 0.8261053562164307, "reward_total_composite_std": 0.1501709669828415} {"timestamp_utc": "2026-04-13T01:41:29Z", "mode": "train", "global_step": 1353, "epoch": 0.13591160220994475, "loss": -0.0523, "grad_norm": 4.680985450744629, "learning_rate": 5.903030303030304e-06, "num_tokens": 2425496.0, "completions/mean_length": 98.5, "completions/min_length": 36.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 39.42857360839844, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.7060085535049438, "rewards/meter/std": 0.42569273710250854, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9604997634887695, "rewards/repeat_soft/std": 0.03425631299614906, "rewards/judge_quality/mean": 0.36124998331069946, "rewards/judge_quality/std": 0.16092921793460846, "rewards/total_composite/mean": 0.5536071062088013, "rewards/total_composite/std": 0.37271034717559814, "reward": 0.5536071062088013, "reward_std": 0.37271037697792053, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1612015664577484, "sampling/sampling_logp_difference/max": 2.1780481338500977, "sampling/importance_sampling_ratio/min": 0.11326239258050919, "sampling/importance_sampling_ratio/mean": 1.0238242149353027, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9070653393864632, "clip_ratio/low_mean": 0.050852272659540176, "clip_ratio/low_min": 0.050852272659540176, "clip_ratio/high_mean": 0.08911602105945349, "clip_ratio/high_max": 0.08911602105945349, "clip_ratio/region_mean": 0.13996829371899366, "reward_total_mean": 0.5536071062088013, "reward_meter_mean": 0.7060085535049438, "reward_meter_std": 0.42569273710250854, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9604997634887695, "reward_repeat_soft_std": 0.03425631299614906, "reward_judge_quality_mean": 0.36124998331069946, "reward_judge_quality_std": 0.16092921793460846, "reward_total_composite_mean": 0.5536071062088013, "reward_total_composite_std": 0.37271034717559814} {"timestamp_utc": "2026-04-13T01:41:41Z", "mode": "train", "global_step": 1354, "epoch": 0.13601205424409843, "loss": -0.0884, "grad_norm": 3.3501336574554443, "learning_rate": 5.9e-06, "num_tokens": 2427067.0, "completions/mean_length": 96.375, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 37.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.6134337186813354, "rewards/meter/std": 0.48660337924957275, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9738596081733704, "rewards/repeat_soft/std": 0.02094823680818081, "rewards/judge_quality/mean": 0.5024999976158142, "rewards/judge_quality/std": 0.2886792719364166, "rewards/total_composite/mean": 0.6410545110702515, "rewards/total_composite/std": 0.34107881784439087, "reward": 0.6410545110702515, "reward_std": 0.34107881784439087, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11328604072332382, "sampling/sampling_logp_difference/max": 1.238179326057434, "sampling/importance_sampling_ratio/min": 0.28991156816482544, "sampling/importance_sampling_ratio/mean": 1.0020400285720825, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6120880357921124, "clip_ratio/low_mean": 0.010731319664046168, "clip_ratio/low_min": 0.010731319664046168, "clip_ratio/high_mean": 0.06647578673437238, "clip_ratio/high_max": 0.06647578673437238, "clip_ratio/region_mean": 0.07720710639841855, "reward_total_mean": 0.6410545110702515, "reward_meter_mean": 0.6134337186813354, "reward_meter_std": 0.48660337924957275, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9738596081733704, "reward_repeat_soft_std": 0.02094823680818081, "reward_judge_quality_mean": 0.5024999976158142, "reward_judge_quality_std": 0.2886792719364166, "reward_total_composite_mean": 0.6410545110702515, "reward_total_composite_std": 0.34107881784439087} {"timestamp_utc": "2026-04-13T01:41:48Z", "mode": "train", "global_step": 1355, "epoch": 0.13611250627825214, "loss": 0.3662, "grad_norm": 11.530550003051758, "learning_rate": 5.8969696969696975e-06, "num_tokens": 2428601.0, "completions/mean_length": 35.75, "completions/min_length": 18.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.75, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.9753167629241943, "rewards/meter/std": 0.045821499079465866, "rewards/count_adherence/mean": 0.375, "rewards/count_adherence/std": 0.5175492167472839, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8926540613174438, "rewards/repeat_soft/std": 0.05331975594162941, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7126579284667969, "rewards/total_composite/std": 0.0707898885011673, "reward": 0.7126579284667969, "reward_std": 0.0707898736000061, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13735948503017426, "sampling/sampling_logp_difference/max": 3.2946701049804688, "sampling/importance_sampling_ratio/min": 0.037080276757478714, "sampling/importance_sampling_ratio/mean": 1.003219485282898, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9496511220932007, "clip_ratio/low_mean": 0.052453776355832815, "clip_ratio/low_min": 0.052453776355832815, "clip_ratio/high_mean": 0.09890351071953773, "clip_ratio/high_max": 0.09890351071953773, "clip_ratio/region_mean": 0.15135728707537055, "reward_total_mean": 0.7126579284667969, "reward_meter_mean": 0.9753167629241943, "reward_meter_std": 0.045821499079465866, "reward_count_adherence_mean": 0.375, "reward_count_adherence_std": 0.5175492167472839, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8926540613174438, "reward_repeat_soft_std": 0.05331975594162941, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7126579284667969, "reward_total_composite_std": 0.0707898885011673} {"timestamp_utc": "2026-04-13T01:41:55Z", "mode": "train", "global_step": 1356, "epoch": 0.13621295831240582, "loss": 0.0171, "grad_norm": 14.746010780334473, "learning_rate": 5.893939393939394e-06, "num_tokens": 2430222.0, "completions/mean_length": 42.625, "completions/min_length": 35.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.625, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.6345236897468567, "rewards/meter/std": 0.3334985375404358, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9410363435745239, "rewards/repeat_soft/std": 0.06094127893447876, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.25587037205696106, "rewards/total_composite/mean": 0.7130143046379089, "rewards/total_composite/std": 0.1548323631286621, "reward": 0.7130143046379089, "reward_std": 0.15483234822750092, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13467665016651154, "sampling/sampling_logp_difference/max": 1.2554941177368164, "sampling/importance_sampling_ratio/min": 0.3207275867462158, "sampling/importance_sampling_ratio/mean": 1.031354546546936, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.934134416282177, "clip_ratio/low_mean": 0.058605073019862175, "clip_ratio/low_min": 0.058605073019862175, "clip_ratio/high_mean": 0.07884379662573338, "clip_ratio/high_max": 0.07884379662573338, "clip_ratio/region_mean": 0.13744886964559555, "reward_total_mean": 0.7130143046379089, "reward_meter_mean": 0.6345236897468567, "reward_meter_std": 0.3334985375404358, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9410363435745239, "reward_repeat_soft_std": 0.06094127893447876, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.25587037205696106, "reward_total_composite_mean": 0.7130143046379089, "reward_total_composite_std": 0.1548323631286621} {"timestamp_utc": "2026-04-13T01:42:02Z", "mode": "train", "global_step": 1357, "epoch": 0.13631341034655953, "loss": 0.0028, "grad_norm": 11.722640991210938, "learning_rate": 5.890909090909091e-06, "num_tokens": 2432426.0, "completions/mean_length": 79.5, "completions/min_length": 68.0, "completions/max_length": 89.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.5, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 89.0, "rewards/meter/mean": 0.741184413433075, "rewards/meter/std": 0.30512428283691406, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8877936601638794, "rewards/repeat_soft/std": 0.05448451638221741, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.675437331199646, "rewards/total_composite/std": 0.13768377900123596, "reward": 0.675437331199646, "reward_std": 0.13768377900123596, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1228928416967392, "sampling/sampling_logp_difference/max": 2.274345874786377, "sampling/importance_sampling_ratio/min": 0.10286417603492737, "sampling/importance_sampling_ratio/mean": 1.0290321111679077, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9978313595056534, "clip_ratio/low_mean": 0.01134868455119431, "clip_ratio/low_min": 0.01134868455119431, "clip_ratio/high_mean": 0.08069514436647296, "clip_ratio/high_max": 0.08069514436647296, "clip_ratio/region_mean": 0.09204382891766727, "reward_total_mean": 0.675437331199646, "reward_meter_mean": 0.741184413433075, "reward_meter_std": 0.30512428283691406, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8877936601638794, "reward_repeat_soft_std": 0.05448451638221741, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.675437331199646, "reward_total_composite_std": 0.13768377900123596} {"timestamp_utc": "2026-04-13T01:42:14Z", "mode": "train", "global_step": 1358, "epoch": 0.1364138623807132, "loss": -0.187, "grad_norm": 2.1551430225372314, "learning_rate": 5.887878787878788e-06, "num_tokens": 2434520.0, "completions/mean_length": 147.75, "completions/min_length": 84.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 95.71428680419922, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.8441834449768066, "rewards/meter/std": 0.22762298583984375, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7979404926300049, "rewards/repeat_soft/std": 0.13102896511554718, "rewards/judge_quality/mean": 0.3399999737739563, "rewards/judge_quality/std": 0.15052290260791779, "rewards/total_composite/mean": 0.6451724767684937, "rewards/total_composite/std": 0.2753659188747406, "reward": 0.6451724767684937, "reward_std": 0.2753659188747406, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10619427263736725, "sampling/sampling_logp_difference/max": 1.2497777938842773, "sampling/importance_sampling_ratio/min": 0.28656846284866333, "sampling/importance_sampling_ratio/mean": 1.0144643783569336, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7407912239432335, "clip_ratio/low_mean": 0.00561797758564353, "clip_ratio/low_min": 0.00561797758564353, "clip_ratio/high_mean": 0.07626903057098389, "clip_ratio/high_max": 0.07626903057098389, "clip_ratio/region_mean": 0.08188700815662742, "reward_total_mean": 0.6451724767684937, "reward_meter_mean": 0.8441834449768066, "reward_meter_std": 0.22762298583984375, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7979404926300049, "reward_repeat_soft_std": 0.13102896511554718, "reward_judge_quality_mean": 0.3399999737739563, "reward_judge_quality_std": 0.15052290260791779, "reward_total_composite_mean": 0.6451724767684937, "reward_total_composite_std": 0.2753659188747406} {"timestamp_utc": "2026-04-13T01:42:21Z", "mode": "train", "global_step": 1359, "epoch": 0.1365143144148669, "loss": 0.0127, "grad_norm": 13.377622604370117, "learning_rate": 5.884848484848486e-06, "num_tokens": 2435805.0, "completions/mean_length": 22.625, "completions/min_length": 21.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.625, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.9953004121780396, "rewards/meter/std": 0.0022072831634432077, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9544244408607483, "rewards/repeat_soft/std": 0.015371029265224934, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8215776681900024, "rewards/total_composite/std": 0.003918899688869715, "reward": 0.8215776681900024, "reward_std": 0.003918912727385759, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13640433549880981, "sampling/sampling_logp_difference/max": 1.1030759811401367, "sampling/importance_sampling_ratio/min": 0.33184874057769775, "sampling/importance_sampling_ratio/mean": 0.9965359568595886, "sampling/importance_sampling_ratio/max": 1.9262287616729736, "entropy": 0.9065387025475502, "clip_ratio/low_mean": 0.05916004395112395, "clip_ratio/low_min": 0.05916004395112395, "clip_ratio/high_mean": 0.07272256724536419, "clip_ratio/high_max": 0.07272256724536419, "clip_ratio/region_mean": 0.13188261119648814, "reward_total_mean": 0.8215776681900024, "reward_meter_mean": 0.9953004121780396, "reward_meter_std": 0.0022072831634432077, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9544244408607483, "reward_repeat_soft_std": 0.015371029265224934, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8215776681900024, "reward_total_composite_std": 0.003918899688869715} {"timestamp_utc": "2026-04-13T01:42:33Z", "mode": "train", "global_step": 1360, "epoch": 0.1366147664490206, "loss": -0.0855, "grad_norm": 2.4182465076446533, "learning_rate": 5.881818181818182e-06, "num_tokens": 2437240.0, "completions/mean_length": 93.375, "completions/min_length": 29.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 33.57143020629883, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.782137393951416, "rewards/meter/std": 0.22362284362316132, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.985435962677002, "rewards/repeat_soft/std": 0.021545695140957832, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.3147788643836975, "rewards/total_composite/mean": 0.6686336994171143, "rewards/total_composite/std": 0.28947290778160095, "reward": 0.6686336994171143, "reward_std": 0.28947290778160095, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16782209277153015, "sampling/sampling_logp_difference/max": 2.5407564640045166, "sampling/importance_sampling_ratio/min": 0.07880676537752151, "sampling/importance_sampling_ratio/mean": 1.007483959197998, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5811209045350552, "clip_ratio/low_mean": 0.0173611119389534, "clip_ratio/low_min": 0.0173611119389534, "clip_ratio/high_mean": 0.1206644894555211, "clip_ratio/high_max": 0.1206644894555211, "clip_ratio/region_mean": 0.1380256013944745, "reward_total_mean": 0.6686336994171143, "reward_meter_mean": 0.782137393951416, "reward_meter_std": 0.22362284362316132, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.985435962677002, "reward_repeat_soft_std": 0.021545695140957832, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.3147788643836975, "reward_total_composite_mean": 0.6686336994171143, "reward_total_composite_std": 0.28947290778160095} {"timestamp_utc": "2026-04-13T01:42:40Z", "mode": "train", "global_step": 1361, "epoch": 0.13671521848317428, "loss": 0.0116, "grad_norm": 6.447414875030518, "learning_rate": 5.878787878787879e-06, "num_tokens": 2439240.0, "completions/mean_length": 63.0, "completions/min_length": 54.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9418991804122925, "rewards/meter/std": 0.08655675500631332, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6004879474639893, "rewards/repeat_soft/std": 0.0788671225309372, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7407784461975098, "rewards/total_composite/std": 0.03359309956431389, "reward": 0.7407784461975098, "reward_std": 0.03359309211373329, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06498279422521591, "sampling/sampling_logp_difference/max": 2.0721964836120605, "sampling/importance_sampling_ratio/min": 0.12590892612934113, "sampling/importance_sampling_ratio/mean": 1.0112518072128296, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3862963914871216, "clip_ratio/low_mean": 0.028293674811720848, "clip_ratio/low_min": 0.028293674811720848, "clip_ratio/high_mean": 0.04864437971264124, "clip_ratio/high_max": 0.04864437971264124, "clip_ratio/region_mean": 0.07693805452436209, "reward_total_mean": 0.7407784461975098, "reward_meter_mean": 0.9418991804122925, "reward_meter_std": 0.08655675500631332, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6004879474639893, "reward_repeat_soft_std": 0.0788671225309372, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7407784461975098, "reward_total_composite_std": 0.03359309956431389} {"timestamp_utc": "2026-04-13T01:42:46Z", "mode": "train", "global_step": 1362, "epoch": 0.136815670517328, "loss": 0.0258, "grad_norm": 15.66557502746582, "learning_rate": 5.875757575757576e-06, "num_tokens": 2440630.0, "completions/mean_length": 22.75, "completions/min_length": 21.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.75, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.7407068014144897, "rewards/meter/std": 0.44852012395858765, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.6826930046081543, "rewards/total_composite/std": 0.19045673310756683, "reward": 0.6826930046081543, "reward_std": 0.19045674800872803, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14067292213439941, "sampling/sampling_logp_difference/max": 1.6444096565246582, "sampling/importance_sampling_ratio/min": 0.19312654435634613, "sampling/importance_sampling_ratio/mean": 0.9971358776092529, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9762941747903824, "clip_ratio/low_mean": 0.027529762126505375, "clip_ratio/low_min": 0.027529762126505375, "clip_ratio/high_mean": 0.10814819345250726, "clip_ratio/high_max": 0.10814819345250726, "clip_ratio/region_mean": 0.13567795557901263, "reward_total_mean": 0.6826930046081543, "reward_meter_mean": 0.7407068014144897, "reward_meter_std": 0.44852012395858765, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.6826930046081543, "reward_total_composite_std": 0.19045673310756683} {"timestamp_utc": "2026-04-13T01:42:58Z", "mode": "train", "global_step": 1363, "epoch": 0.13691612255148167, "loss": -0.3012, "grad_norm": 1.3076545000076294, "learning_rate": 5.872727272727273e-06, "num_tokens": 2443010.0, "completions/mean_length": 452.5, "completions/min_length": 323.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.625, "completions/mean_terminated_length": 353.3333435058594, "completions/min_terminated_length": 323.0, "completions/max_terminated_length": 382.0, "rewards/meter/mean": 0.7111875414848328, "rewards/meter/std": 0.3169199228286743, "rewards/count_adherence/mean": 0.0625, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.7258596420288086, "rewards/repeat_soft/std": 0.199764221906662, "rewards/judge_quality/mean": 0.12125000357627869, "rewards/judge_quality/std": 0.08806450664997101, "rewards/total_composite/mean": 0.28930050134658813, "rewards/total_composite/std": 0.2664940655231476, "reward": 0.28930050134658813, "reward_std": 0.2664940655231476, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05777793750166893, "sampling/sampling_logp_difference/max": 1.1492958068847656, "sampling/importance_sampling_ratio/min": 0.31685981154441833, "sampling/importance_sampling_ratio/mean": 1.002070665359497, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.16243772581219673, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.02161393268033862, "clip_ratio/high_max": 0.02161393268033862, "clip_ratio/region_mean": 0.02161393268033862, "reward_total_mean": 0.28930050134658813, "reward_meter_mean": 0.7111875414848328, "reward_meter_std": 0.3169199228286743, "reward_count_adherence_mean": 0.0625, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.7258596420288086, "reward_repeat_soft_std": 0.199764221906662, "reward_judge_quality_mean": 0.12125000357627869, "reward_judge_quality_std": 0.08806450664997101, "reward_total_composite_mean": 0.28930050134658813, "reward_total_composite_std": 0.2664940655231476} {"timestamp_utc": "2026-04-13T01:43:10Z", "mode": "train", "global_step": 1364, "epoch": 0.13701657458563535, "loss": -0.1165, "grad_norm": 2.2662265300750732, "learning_rate": 5.8696969696969694e-06, "num_tokens": 2444512.0, "completions/mean_length": 99.75, "completions/min_length": 39.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 40.85714340209961, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.7656885385513306, "rewards/meter/std": 0.3116970956325531, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9134343266487122, "rewards/repeat_soft/std": 0.0584520623087883, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.23439893126487732, "rewards/total_composite/mean": 0.6735928058624268, "rewards/total_composite/std": 0.29891255497932434, "reward": 0.6735928058624268, "reward_std": 0.29891255497932434, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11832264065742493, "sampling/sampling_logp_difference/max": 1.8087468147277832, "sampling/importance_sampling_ratio/min": 0.16385935246944427, "sampling/importance_sampling_ratio/mean": 1.0294840335845947, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7923390567302704, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06963638076558709, "clip_ratio/high_max": 0.06963638076558709, "clip_ratio/region_mean": 0.06963638076558709, "reward_total_mean": 0.6735928058624268, "reward_meter_mean": 0.7656885385513306, "reward_meter_std": 0.3116970956325531, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9134343266487122, "reward_repeat_soft_std": 0.0584520623087883, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.23439893126487732, "reward_total_composite_mean": 0.6735928058624268, "reward_total_composite_std": 0.29891255497932434} {"timestamp_utc": "2026-04-13T01:43:17Z", "mode": "train", "global_step": 1365, "epoch": 0.13711702661978906, "loss": -0.006, "grad_norm": 11.46400260925293, "learning_rate": 5.8666666666666675e-06, "num_tokens": 2446110.0, "completions/mean_length": 38.75, "completions/min_length": 33.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.8857302665710449, "rewards/meter/std": 0.18392860889434814, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7760871648788452, "rewards/repeat_soft/std": 0.094293013215065, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.12631450593471527, "rewards/total_composite/mean": 0.725562334060669, "rewards/total_composite/std": 0.10348951071500778, "reward": 0.725562334060669, "reward_std": 0.10348950326442719, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11549367755651474, "sampling/sampling_logp_difference/max": 1.6490581035614014, "sampling/importance_sampling_ratio/min": 0.19223088026046753, "sampling/importance_sampling_ratio/mean": 1.0113184452056885, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7318059764802456, "clip_ratio/low_mean": 0.03853345010429621, "clip_ratio/low_min": 0.03853345010429621, "clip_ratio/high_mean": 0.10378486034460366, "clip_ratio/high_max": 0.10378486034460366, "clip_ratio/region_mean": 0.14231831044889987, "reward_total_mean": 0.725562334060669, "reward_meter_mean": 0.8857302665710449, "reward_meter_std": 0.18392860889434814, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7760871648788452, "reward_repeat_soft_std": 0.094293013215065, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.12631450593471527, "reward_total_composite_mean": 0.725562334060669, "reward_total_composite_std": 0.10348951071500778} {"timestamp_utc": "2026-04-13T01:43:29Z", "mode": "train", "global_step": 1366, "epoch": 0.13721747865394274, "loss": 0.2825, "grad_norm": 3.671198606491089, "learning_rate": 5.863636363636364e-06, "num_tokens": 2448710.0, "completions/mean_length": 241.0, "completions/min_length": 57.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 202.2857208251953, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 393.0, "rewards/meter/mean": 0.8731129169464111, "rewards/meter/std": 0.17524968087673187, "rewards/count_adherence/mean": 0.1875, "rewards/count_adherence/std": 0.25877460837364197, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7316066026687622, "rewards/repeat_soft/std": 0.20608854293823242, "rewards/judge_quality/mean": 0.29749998450279236, "rewards/judge_quality/std": 0.1537855714559555, "rewards/total_composite/mean": 0.5322076082229614, "rewards/total_composite/std": 0.24679110944271088, "reward": 0.5322076082229614, "reward_std": 0.24679109454154968, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07861162722110748, "sampling/sampling_logp_difference/max": 1.552572250366211, "sampling/importance_sampling_ratio/min": 0.21170271933078766, "sampling/importance_sampling_ratio/mean": 1.0208165645599365, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7916410453617573, "clip_ratio/low_mean": 0.0095331990160048, "clip_ratio/low_min": 0.0095331990160048, "clip_ratio/high_mean": 0.06579802185297012, "clip_ratio/high_max": 0.06579802185297012, "clip_ratio/region_mean": 0.07533122086897492, "reward_total_mean": 0.5322076082229614, "reward_meter_mean": 0.8731129169464111, "reward_meter_std": 0.17524968087673187, "reward_count_adherence_mean": 0.1875, "reward_count_adherence_std": 0.25877460837364197, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7316066026687622, "reward_repeat_soft_std": 0.20608854293823242, "reward_judge_quality_mean": 0.29749998450279236, "reward_judge_quality_std": 0.1537855714559555, "reward_total_composite_mean": 0.5322076082229614, "reward_total_composite_std": 0.24679110944271088} {"timestamp_utc": "2026-04-13T01:43:36Z", "mode": "train", "global_step": 1367, "epoch": 0.13731793068809645, "loss": 0.0036, "grad_norm": 5.639797210693359, "learning_rate": 5.860606060606061e-06, "num_tokens": 2450853.0, "completions/mean_length": 96.875, "completions/min_length": 92.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.875, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.9118024110794067, "rewards/meter/std": 0.20018433034420013, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7580682039260864, "rewards/repeat_soft/std": 0.10958189517259598, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7585554122924805, "rewards/total_composite/std": 0.09243952482938766, "reward": 0.7585554122924805, "reward_std": 0.09243950992822647, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09127871692180634, "sampling/sampling_logp_difference/max": 1.390432596206665, "sampling/importance_sampling_ratio/min": 0.24896757304668427, "sampling/importance_sampling_ratio/mean": 1.00868558883667, "sampling/importance_sampling_ratio/max": 1.8227721452713013, "entropy": 0.686897374689579, "clip_ratio/low_mean": 0.003989361692219973, "clip_ratio/low_min": 0.003989361692219973, "clip_ratio/high_mean": 0.07863155007362366, "clip_ratio/high_max": 0.07863155007362366, "clip_ratio/region_mean": 0.08262091176584363, "reward_total_mean": 0.7585554122924805, "reward_meter_mean": 0.9118024110794067, "reward_meter_std": 0.20018433034420013, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7580682039260864, "reward_repeat_soft_std": 0.10958189517259598, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7585554122924805, "reward_total_composite_std": 0.09243952482938766} {"timestamp_utc": "2026-04-13T01:43:42Z", "mode": "train", "global_step": 1368, "epoch": 0.13741838272225013, "loss": 0.034, "grad_norm": 15.558755874633789, "learning_rate": 5.8575757575757584e-06, "num_tokens": 2452511.0, "completions/mean_length": 35.25, "completions/min_length": 32.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.25, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.5346692800521851, "rewards/meter/std": 0.4472871422767639, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.929846465587616, "rewards/repeat_soft/std": 0.07309762388467789, "rewards/judge_quality/mean": 0.38874998688697815, "rewards/judge_quality/std": 0.08675704896450043, "rewards/total_composite/mean": 0.6002108454704285, "rewards/total_composite/std": 0.19882752001285553, "reward": 0.6002108454704285, "reward_std": 0.19882753491401672, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1708817183971405, "sampling/sampling_logp_difference/max": 1.4967894554138184, "sampling/importance_sampling_ratio/min": 0.22384768724441528, "sampling/importance_sampling_ratio/mean": 1.0163028240203857, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9803749583661556, "clip_ratio/low_mean": 0.09077381156384945, "clip_ratio/low_min": 0.09077381156384945, "clip_ratio/high_mean": 0.07126304134726524, "clip_ratio/high_max": 0.07126304134726524, "clip_ratio/region_mean": 0.1620368529111147, "reward_total_mean": 0.6002108454704285, "reward_meter_mean": 0.5346692800521851, "reward_meter_std": 0.4472871422767639, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.929846465587616, "reward_repeat_soft_std": 0.07309762388467789, "reward_judge_quality_mean": 0.38874998688697815, "reward_judge_quality_std": 0.08675704896450043, "reward_total_composite_mean": 0.6002108454704285, "reward_total_composite_std": 0.19882752001285553} {"timestamp_utc": "2026-04-13T01:43:50Z", "mode": "train", "global_step": 1369, "epoch": 0.1375188347564038, "loss": 0.0413, "grad_norm": 8.824570655822754, "learning_rate": 5.854545454545455e-06, "num_tokens": 2454454.0, "completions/mean_length": 75.875, "completions/min_length": 59.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.875, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.8366955518722534, "rewards/meter/std": 0.32982441782951355, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7845086455345154, "rewards/repeat_soft/std": 0.11270208656787872, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7011513710021973, "rewards/total_composite/std": 0.15256915986537933, "reward": 0.7011513710021973, "reward_std": 0.15256913006305695, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09266103059053421, "sampling/sampling_logp_difference/max": 1.3347692489624023, "sampling/importance_sampling_ratio/min": 0.2632189095020294, "sampling/importance_sampling_ratio/mean": 1.0087474584579468, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4982188865542412, "clip_ratio/low_mean": 0.011904762126505375, "clip_ratio/low_min": 0.011904762126505375, "clip_ratio/high_mean": 0.08162437099963427, "clip_ratio/high_max": 0.08162437099963427, "clip_ratio/region_mean": 0.09352913312613964, "reward_total_mean": 0.7011513710021973, "reward_meter_mean": 0.8366955518722534, "reward_meter_std": 0.32982441782951355, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7845086455345154, "reward_repeat_soft_std": 0.11270208656787872, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7011513710021973, "reward_total_composite_std": 0.15256915986537933} {"timestamp_utc": "2026-04-13T01:44:02Z", "mode": "train", "global_step": 1370, "epoch": 0.13761928679055752, "loss": -0.1546, "grad_norm": 1.6761891841888428, "learning_rate": 5.851515151515152e-06, "num_tokens": 2456251.0, "completions/mean_length": 123.625, "completions/min_length": 53.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 68.14286041259766, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.9431145787239075, "rewards/meter/std": 0.10813675820827484, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.33034372329711914, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6328310966491699, "rewards/repeat_soft/std": 0.13900338113307953, "rewards/judge_quality/mean": 0.2799999713897705, "rewards/judge_quality/std": 0.09304375946521759, "rewards/total_composite/mean": 0.6779346466064453, "rewards/total_composite/std": 0.11838347464799881, "reward": 0.6779346466064453, "reward_std": 0.11838346719741821, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09583054482936859, "sampling/sampling_logp_difference/max": 1.296797275543213, "sampling/importance_sampling_ratio/min": 0.2734060287475586, "sampling/importance_sampling_ratio/mean": 1.0098857879638672, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5641437694430351, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.07250200817361474, "clip_ratio/high_max": 0.07250200817361474, "clip_ratio/region_mean": 0.07250200817361474, "reward_total_mean": 0.6779346466064453, "reward_meter_mean": 0.9431145787239075, "reward_meter_std": 0.10813675820827484, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.33034372329711914, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6328310966491699, "reward_repeat_soft_std": 0.13900338113307953, "reward_judge_quality_mean": 0.2799999713897705, "reward_judge_quality_std": 0.09304375946521759, "reward_total_composite_mean": 0.6779346466064453, "reward_total_composite_std": 0.11838347464799881} {"timestamp_utc": "2026-04-13T01:44:09Z", "mode": "train", "global_step": 1371, "epoch": 0.1377197388247112, "loss": -0.0101, "grad_norm": 14.216009140014648, "learning_rate": 5.8484848484848485e-06, "num_tokens": 2457637.0, "completions/mean_length": 24.25, "completions/min_length": 23.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.25, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.9828266501426697, "rewards/meter/std": 0.009167668409645557, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8971240520477295, "rewards/repeat_soft/std": 0.16015005111694336, "rewards/judge_quality/mean": 0.3125, "rewards/judge_quality/std": 0.11877349019050598, "rewards/total_composite/mean": 0.7757344245910645, "rewards/total_composite/std": 0.031574174761772156, "reward": 0.7757344245910645, "reward_std": 0.031574174761772156, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13108013570308685, "sampling/sampling_logp_difference/max": 1.5669631958007812, "sampling/importance_sampling_ratio/min": 0.20867793262004852, "sampling/importance_sampling_ratio/mean": 1.0253901481628418, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1258932501077652, "clip_ratio/low_mean": 0.036354515701532364, "clip_ratio/low_min": 0.036354515701532364, "clip_ratio/high_mean": 0.019615384750068188, "clip_ratio/high_max": 0.019615384750068188, "clip_ratio/region_mean": 0.05596990045160055, "reward_total_mean": 0.7757344245910645, "reward_meter_mean": 0.9828266501426697, "reward_meter_std": 0.009167668409645557, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8971240520477295, "reward_repeat_soft_std": 0.16015005111694336, "reward_judge_quality_mean": 0.3125, "reward_judge_quality_std": 0.11877349019050598, "reward_total_composite_mean": 0.7757344245910645, "reward_total_composite_std": 0.031574174761772156} {"timestamp_utc": "2026-04-13T01:44:16Z", "mode": "train", "global_step": 1372, "epoch": 0.1378201908588649, "loss": 0.0142, "grad_norm": 12.79295539855957, "learning_rate": 5.845454545454547e-06, "num_tokens": 2459391.0, "completions/mean_length": 46.25, "completions/min_length": 34.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.25, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.7806754112243652, "rewards/meter/std": 0.28153255581855774, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9428256750106812, "rewards/repeat_soft/std": 0.03568185865879059, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.6837115287780762, "rewards/total_composite/std": 0.12898629903793335, "reward": 0.6837115287780762, "reward_std": 0.12898628413677216, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1489105224609375, "sampling/sampling_logp_difference/max": 1.6136369705200195, "sampling/importance_sampling_ratio/min": 0.19916194677352905, "sampling/importance_sampling_ratio/mean": 1.0245641469955444, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1247057616710663, "clip_ratio/low_mean": 0.06613130681216717, "clip_ratio/low_min": 0.06613130681216717, "clip_ratio/high_mean": 0.05549857905134559, "clip_ratio/high_max": 0.05549857905134559, "clip_ratio/region_mean": 0.12162988586351275, "reward_total_mean": 0.6837115287780762, "reward_meter_mean": 0.7806754112243652, "reward_meter_std": 0.28153255581855774, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9428256750106812, "reward_repeat_soft_std": 0.03568185865879059, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.6837115287780762, "reward_total_composite_std": 0.12898629903793335} {"timestamp_utc": "2026-04-13T01:44:28Z", "mode": "train", "global_step": 1373, "epoch": 0.13792064289301859, "loss": -0.1835, "grad_norm": 1.0940401554107666, "learning_rate": 5.842424242424243e-06, "num_tokens": 2461437.0, "completions/mean_length": 142.75, "completions/min_length": 78.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 90.00000762939453, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.889041543006897, "rewards/meter/std": 0.2508357763290405, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.2651650309562683, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.5496461987495422, "rewards/repeat_soft/std": 0.19990971684455872, "rewards/judge_quality/mean": 0.2762500047683716, "rewards/judge_quality/std": 0.13689802587032318, "rewards/total_composite/mean": 0.6307769417762756, "rewards/total_composite/std": 0.25851890444755554, "reward": 0.6307769417762756, "reward_std": 0.25851890444755554, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05099700391292572, "sampling/sampling_logp_difference/max": 1.3290960788726807, "sampling/importance_sampling_ratio/min": 0.2647164463996887, "sampling/importance_sampling_ratio/mean": 1.003748893737793, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2415849706158042, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.04495816840790212, "clip_ratio/high_max": 0.04495816840790212, "clip_ratio/region_mean": 0.04495816840790212, "reward_total_mean": 0.6307769417762756, "reward_meter_mean": 0.889041543006897, "reward_meter_std": 0.2508357763290405, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.2651650309562683, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.5496461987495422, "reward_repeat_soft_std": 0.19990971684455872, "reward_judge_quality_mean": 0.2762500047683716, "reward_judge_quality_std": 0.13689802587032318, "reward_total_composite_mean": 0.6307769417762756, "reward_total_composite_std": 0.25851890444755554} {"timestamp_utc": "2026-04-13T01:44:40Z", "mode": "train", "global_step": 1374, "epoch": 0.13802109492717227, "loss": -0.1396, "grad_norm": 3.852116107940674, "learning_rate": 5.83939393939394e-06, "num_tokens": 2463492.0, "completions/mean_length": 148.875, "completions/min_length": 77.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 97.00000762939453, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.4032798707485199, "rewards/meter/std": 0.3241139054298401, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.29124119877815247, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8200811743736267, "rewards/repeat_soft/std": 0.1204463392496109, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.48248404264450073, "rewards/total_composite/std": 0.19951559603214264, "reward": 0.48248404264450073, "reward_std": 0.19951556622982025, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09980063885450363, "sampling/sampling_logp_difference/max": 2.0611681938171387, "sampling/importance_sampling_ratio/min": 0.12730516493320465, "sampling/importance_sampling_ratio/mean": 1.0031243562698364, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40615861117839813, "clip_ratio/low_mean": 0.029310510493814945, "clip_ratio/low_min": 0.029310510493814945, "clip_ratio/high_mean": 0.04846938932314515, "clip_ratio/high_max": 0.04846938932314515, "clip_ratio/region_mean": 0.0777798998169601, "reward_total_mean": 0.48248404264450073, "reward_meter_mean": 0.4032798707485199, "reward_meter_std": 0.3241139054298401, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.29124119877815247, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8200811743736267, "reward_repeat_soft_std": 0.1204463392496109, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.48248404264450073, "reward_total_composite_std": 0.19951559603214264} {"timestamp_utc": "2026-04-13T01:44:51Z", "mode": "train", "global_step": 1375, "epoch": 0.13812154696132597, "loss": -0.1264, "grad_norm": 2.703648328781128, "learning_rate": 5.836363636363637e-06, "num_tokens": 2465317.0, "completions/mean_length": 129.125, "completions/min_length": 69.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 74.42857360839844, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.9575926661491394, "rewards/meter/std": 0.06832104921340942, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.4987461268901825, "rewards/repeat_soft/std": 0.17331312596797943, "rewards/judge_quality/mean": 0.2762500047683716, "rewards/judge_quality/std": 0.12603145837783813, "rewards/total_composite/mean": 0.7042913436889648, "rewards/total_composite/std": 0.07780779898166656, "reward": 0.7042913436889648, "reward_std": 0.07780779153108597, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05577255040407181, "sampling/sampling_logp_difference/max": 1.3455884456634521, "sampling/importance_sampling_ratio/min": 0.2603864371776581, "sampling/importance_sampling_ratio/mean": 1.0095852613449097, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2547594178467989, "clip_ratio/low_mean": 0.004893346689641476, "clip_ratio/low_min": 0.004893346689641476, "clip_ratio/high_mean": 0.01575100771151483, "clip_ratio/high_max": 0.01575100771151483, "clip_ratio/region_mean": 0.020644354401156306, "reward_total_mean": 0.7042913436889648, "reward_meter_mean": 0.9575926661491394, "reward_meter_std": 0.06832104921340942, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.4987461268901825, "reward_repeat_soft_std": 0.17331312596797943, "reward_judge_quality_mean": 0.2762500047683716, "reward_judge_quality_std": 0.12603145837783813, "reward_total_composite_mean": 0.7042913436889648, "reward_total_composite_std": 0.07780779898166656} {"timestamp_utc": "2026-04-13T01:44:58Z", "mode": "train", "global_step": 1376, "epoch": 0.13822199899547966, "loss": -0.0256, "grad_norm": 11.935321807861328, "learning_rate": 5.833333333333334e-06, "num_tokens": 2466924.0, "completions/mean_length": 38.875, "completions/min_length": 33.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.9450453519821167, "rewards/meter/std": 0.12241283804178238, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8412350416183472, "rewards/repeat_soft/std": 0.1525059938430786, "rewards/judge_quality/mean": 0.3725000023841858, "rewards/judge_quality/std": 0.11055056750774384, "rewards/total_composite/mean": 0.771143913269043, "rewards/total_composite/std": 0.0663236454129219, "reward": 0.771143913269043, "reward_std": 0.0663236603140831, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11439573019742966, "sampling/sampling_logp_difference/max": 1.1463909149169922, "sampling/importance_sampling_ratio/min": 0.31778159737586975, "sampling/importance_sampling_ratio/mean": 1.0276464223861694, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.813943475484848, "clip_ratio/low_mean": 0.02722902176901698, "clip_ratio/low_min": 0.02722902176901698, "clip_ratio/high_mean": 0.0708274426870048, "clip_ratio/high_max": 0.0708274426870048, "clip_ratio/region_mean": 0.09805646445602179, "reward_total_mean": 0.771143913269043, "reward_meter_mean": 0.9450453519821167, "reward_meter_std": 0.12241283804178238, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8412350416183472, "reward_repeat_soft_std": 0.1525059938430786, "reward_judge_quality_mean": 0.3725000023841858, "reward_judge_quality_std": 0.11055056750774384, "reward_total_composite_mean": 0.771143913269043, "reward_total_composite_std": 0.0663236454129219} {"timestamp_utc": "2026-04-13T01:45:10Z", "mode": "train", "global_step": 1377, "epoch": 0.13832245102963334, "loss": 0.1209, "grad_norm": 3.6274027824401855, "learning_rate": 5.83030303030303e-06, "num_tokens": 2468971.0, "completions/mean_length": 155.875, "completions/min_length": 64.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 105.00000762939453, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 273.0, "rewards/meter/mean": 0.9228025674819946, "rewards/meter/std": 0.11118209362030029, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.40089187026023865, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7892471551895142, "rewards/repeat_soft/std": 0.15058892965316772, "rewards/judge_quality/mean": 0.45250001549720764, "rewards/judge_quality/std": 0.18100909888744354, "rewards/total_composite/mean": 0.7236858606338501, "rewards/total_composite/std": 0.10062537342309952, "reward": 0.7236858606338501, "reward_std": 0.10062538087368011, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08170418441295624, "sampling/sampling_logp_difference/max": 1.4064054489135742, "sampling/importance_sampling_ratio/min": 0.2450224608182907, "sampling/importance_sampling_ratio/mean": 1.0051332712173462, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4798858277499676, "clip_ratio/low_mean": 0.0082417584490031, "clip_ratio/low_min": 0.0082417584490031, "clip_ratio/high_mean": 0.06116462126374245, "clip_ratio/high_max": 0.06116462126374245, "clip_ratio/region_mean": 0.06940637971274555, "reward_total_mean": 0.7236858606338501, "reward_meter_mean": 0.9228025674819946, "reward_meter_std": 0.11118209362030029, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.40089187026023865, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7892471551895142, "reward_repeat_soft_std": 0.15058892965316772, "reward_judge_quality_mean": 0.45250001549720764, "reward_judge_quality_std": 0.18100909888744354, "reward_total_composite_mean": 0.7236858606338501, "reward_total_composite_std": 0.10062537342309952} {"timestamp_utc": "2026-04-13T01:45:17Z", "mode": "train", "global_step": 1378, "epoch": 0.13842290306378705, "loss": 0.0892, "grad_norm": 6.642712116241455, "learning_rate": 5.8272727272727285e-06, "num_tokens": 2471380.0, "completions/mean_length": 119.125, "completions/min_length": 98.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.125, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.9951872825622559, "rewards/meter/std": 0.004394045565277338, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7122833728790283, "rewards/repeat_soft/std": 0.14724230766296387, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.7508126497268677, "rewards/total_composite/std": 0.034532658755779266, "reward": 0.7508126497268677, "reward_std": 0.03453264385461807, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10232824832201004, "sampling/sampling_logp_difference/max": 1.329157829284668, "sampling/importance_sampling_ratio/min": 0.2647000849246979, "sampling/importance_sampling_ratio/mean": 1.0060219764709473, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7393070720136166, "clip_ratio/low_mean": 0.021702975034713745, "clip_ratio/low_min": 0.021702975034713745, "clip_ratio/high_mean": 0.07345864456146955, "clip_ratio/high_max": 0.07345864456146955, "clip_ratio/region_mean": 0.0951616195961833, "reward_total_mean": 0.7508126497268677, "reward_meter_mean": 0.9951872825622559, "reward_meter_std": 0.004394045565277338, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7122833728790283, "reward_repeat_soft_std": 0.14724230766296387, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.7508126497268677, "reward_total_composite_std": 0.034532658755779266} {"timestamp_utc": "2026-04-13T01:45:23Z", "mode": "train", "global_step": 1379, "epoch": 0.13852335509794073, "loss": -0.0147, "grad_norm": 16.54239845275879, "learning_rate": 5.824242424242425e-06, "num_tokens": 2472811.0, "completions/mean_length": 22.875, "completions/min_length": 21.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.875, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.8855236768722534, "rewards/meter/std": 0.26723581552505493, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9584129452705383, "rewards/repeat_soft/std": 0.011559911072254181, "rewards/judge_quality/mean": 0.3424999713897705, "rewards/judge_quality/std": 0.09953462332487106, "rewards/total_composite/mean": 0.7470769286155701, "rewards/total_composite/std": 0.11461164802312851, "reward": 0.7470769286155701, "reward_std": 0.11461161822080612, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1430753469467163, "sampling/sampling_logp_difference/max": 1.676095962524414, "sampling/importance_sampling_ratio/min": 0.1871030032634735, "sampling/importance_sampling_ratio/mean": 0.9819827675819397, "sampling/importance_sampling_ratio/max": 1.736879587173462, "entropy": 0.9006307050585747, "clip_ratio/low_mean": 0.028409091755747795, "clip_ratio/low_min": 0.028409091755747795, "clip_ratio/high_mean": 0.1416326160542667, "clip_ratio/high_max": 0.1416326160542667, "clip_ratio/region_mean": 0.1700417078100145, "reward_total_mean": 0.7470769286155701, "reward_meter_mean": 0.8855236768722534, "reward_meter_std": 0.26723581552505493, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9584129452705383, "reward_repeat_soft_std": 0.011559911072254181, "reward_judge_quality_mean": 0.3424999713897705, "reward_judge_quality_std": 0.09953462332487106, "reward_total_composite_mean": 0.7470769286155701, "reward_total_composite_std": 0.11461164802312851} {"timestamp_utc": "2026-04-13T01:45:35Z", "mode": "train", "global_step": 1380, "epoch": 0.13862380713209443, "loss": 0.0056, "grad_norm": 4.198464870452881, "learning_rate": 5.821212121212122e-06, "num_tokens": 2474927.0, "completions/mean_length": 154.5, "completions/min_length": 93.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 103.42857360839844, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 115.0, "rewards/meter/mean": 0.9770861864089966, "rewards/meter/std": 0.037616144865751266, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8161088228225708, "rewards/repeat_soft/std": 0.1383945643901825, "rewards/judge_quality/mean": 0.2762500047683716, "rewards/judge_quality/std": 0.13689802587032318, "rewards/total_composite/mean": 0.6378435492515564, "rewards/total_composite/std": 0.2680339217185974, "reward": 0.6378435492515564, "reward_std": 0.2680339217185974, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09528955817222595, "sampling/sampling_logp_difference/max": 1.9315202236175537, "sampling/importance_sampling_ratio/min": 0.1449277102947235, "sampling/importance_sampling_ratio/mean": 1.0116859674453735, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.647913433611393, "clip_ratio/low_mean": 0.01304347813129425, "clip_ratio/low_min": 0.01304347813129425, "clip_ratio/high_mean": 0.06610892992466688, "clip_ratio/high_max": 0.06610892992466688, "clip_ratio/region_mean": 0.07915240805596113, "reward_total_mean": 0.6378435492515564, "reward_meter_mean": 0.9770861864089966, "reward_meter_std": 0.037616144865751266, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.2651650309562683, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8161088228225708, "reward_repeat_soft_std": 0.1383945643901825, "reward_judge_quality_mean": 0.2762500047683716, "reward_judge_quality_std": 0.13689802587032318, "reward_total_composite_mean": 0.6378435492515564, "reward_total_composite_std": 0.2680339217185974} {"timestamp_utc": "2026-04-13T01:45:49Z", "mode": "train", "global_step": 1381, "epoch": 0.13872425916624812, "loss": 0.0867, "grad_norm": 5.463994979858398, "learning_rate": 5.8181818181818185e-06, "num_tokens": 2477219.0, "completions/mean_length": 107.5, "completions/min_length": 98.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.5, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.9851288199424744, "rewards/meter/std": 0.00950995646417141, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6894950866699219, "rewards/repeat_soft/std": 0.08892365545034409, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7644450068473816, "rewards/total_composite/std": 0.03722214698791504, "reward": 0.7644450068473816, "reward_std": 0.037222154438495636, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08050207048654556, "sampling/sampling_logp_difference/max": 2.5842833518981934, "sampling/importance_sampling_ratio/min": 0.07545012980699539, "sampling/importance_sampling_ratio/mean": 1.0213428735733032, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6205888576805592, "clip_ratio/low_mean": 0.025017806328833103, "clip_ratio/low_min": 0.025017806328833103, "clip_ratio/high_mean": 0.040763274766504765, "clip_ratio/high_max": 0.040763274766504765, "clip_ratio/region_mean": 0.06578108109533787, "reward_total_mean": 0.7644450068473816, "reward_meter_mean": 0.9851288199424744, "reward_meter_std": 0.00950995646417141, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6894950866699219, "reward_repeat_soft_std": 0.08892365545034409, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7644450068473816, "reward_total_composite_std": 0.03722214698791504} {"timestamp_utc": "2026-04-13T01:46:02Z", "mode": "train", "global_step": 1382, "epoch": 0.1388247112004018, "loss": -0.1553, "grad_norm": 2.8385097980499268, "learning_rate": 5.815151515151516e-06, "num_tokens": 2479129.0, "completions/mean_length": 125.75, "completions/min_length": 64.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 70.5714340209961, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.6200876235961914, "rewards/meter/std": 0.4166737198829651, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.18600596487522125, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8672417998313904, "rewards/repeat_soft/std": 0.06918510794639587, "rewards/judge_quality/mean": 0.5550000071525574, "rewards/judge_quality/std": 0.33393755555152893, "rewards/total_composite/mean": 0.6462858319282532, "rewards/total_composite/std": 0.3038969933986664, "reward": 0.6462858319282532, "reward_std": 0.303896963596344, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07884561270475388, "sampling/sampling_logp_difference/max": 1.6272759437561035, "sampling/importance_sampling_ratio/min": 0.19646401703357697, "sampling/importance_sampling_ratio/mean": 1.0129324197769165, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36961888149380684, "clip_ratio/low_mean": 0.020385206677019596, "clip_ratio/low_min": 0.020385206677019596, "clip_ratio/high_mean": 0.04902395885437727, "clip_ratio/high_max": 0.04902395885437727, "clip_ratio/region_mean": 0.06940916553139687, "reward_total_mean": 0.6462858319282532, "reward_meter_mean": 0.6200876235961914, "reward_meter_std": 0.4166737198829651, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.18600596487522125, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8672417998313904, "reward_repeat_soft_std": 0.06918510794639587, "reward_judge_quality_mean": 0.5550000071525574, "reward_judge_quality_std": 0.33393755555152893, "reward_total_composite_mean": 0.6462858319282532, "reward_total_composite_std": 0.3038969933986664} {"timestamp_utc": "2026-04-13T01:46:09Z", "mode": "train", "global_step": 1383, "epoch": 0.1389251632345555, "loss": 0.0362, "grad_norm": 6.775473117828369, "learning_rate": 5.812121212121212e-06, "num_tokens": 2481164.0, "completions/mean_length": 81.375, "completions/min_length": 71.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.375, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9653944373130798, "rewards/meter/std": 0.03963348641991615, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8470273017883301, "rewards/repeat_soft/std": 0.10441792011260986, "rewards/judge_quality/mean": 0.2887499928474426, "rewards/judge_quality/std": 0.11630470305681229, "rewards/total_composite/mean": 0.7557551860809326, "rewards/total_composite/std": 0.0383811853826046, "reward": 0.7557551860809326, "reward_std": 0.038381192833185196, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09808217734098434, "sampling/sampling_logp_difference/max": 1.3560352325439453, "sampling/importance_sampling_ratio/min": 0.25768041610717773, "sampling/importance_sampling_ratio/mean": 1.0167022943496704, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7127524837851524, "clip_ratio/low_mean": 0.05148072028532624, "clip_ratio/low_min": 0.05148072028532624, "clip_ratio/high_mean": 0.04552819626405835, "clip_ratio/high_max": 0.04552819626405835, "clip_ratio/region_mean": 0.0970089165493846, "reward_total_mean": 0.7557551860809326, "reward_meter_mean": 0.9653944373130798, "reward_meter_std": 0.03963348641991615, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8470273017883301, "reward_repeat_soft_std": 0.10441792011260986, "reward_judge_quality_mean": 0.2887499928474426, "reward_judge_quality_std": 0.11630470305681229, "reward_total_composite_mean": 0.7557551860809326, "reward_total_composite_std": 0.0383811853826046} {"timestamp_utc": "2026-04-13T01:46:16Z", "mode": "train", "global_step": 1384, "epoch": 0.13902561526870919, "loss": 0.0681, "grad_norm": 20.710678100585938, "learning_rate": 5.8090909090909095e-06, "num_tokens": 2482954.0, "completions/mean_length": 68.75, "completions/min_length": 62.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.75, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9532653093338013, "rewards/meter/std": 0.05389257147908211, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9621062874794006, "rewards/repeat_soft/std": 0.029222683981060982, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.8124300241470337, "rewards/total_composite/std": 0.027022775262594223, "reward": 0.8124300241470337, "reward_std": 0.027022775262594223, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09948781877756119, "sampling/sampling_logp_difference/max": 1.7439966201782227, "sampling/importance_sampling_ratio/min": 0.17482030391693115, "sampling/importance_sampling_ratio/mean": 1.003517508506775, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5226207450032234, "clip_ratio/low_mean": 0.01887917355634272, "clip_ratio/low_min": 0.01887917355634272, "clip_ratio/high_mean": 0.0688763982616365, "clip_ratio/high_max": 0.0688763982616365, "clip_ratio/region_mean": 0.08775557181797922, "reward_total_mean": 0.8124300241470337, "reward_meter_mean": 0.9532653093338013, "reward_meter_std": 0.05389257147908211, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9621062874794006, "reward_repeat_soft_std": 0.029222683981060982, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.8124300241470337, "reward_total_composite_std": 0.027022775262594223} {"timestamp_utc": "2026-04-13T01:46:23Z", "mode": "train", "global_step": 1385, "epoch": 0.1391260673028629, "loss": 0.0082, "grad_norm": 11.440997123718262, "learning_rate": 5.806060606060606e-06, "num_tokens": 2484393.0, "completions/mean_length": 32.875, "completions/min_length": 29.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.875, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9542100429534912, "rewards/meter/std": 0.04555569961667061, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8767144083976746, "rewards/repeat_soft/std": 0.1099141389131546, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.11667262762784958, "rewards/total_composite/mean": 0.8054409623146057, "rewards/total_composite/std": 0.04349781945347786, "reward": 0.8054409623146057, "reward_std": 0.043497826904058456, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08631597459316254, "sampling/sampling_logp_difference/max": 1.0098570585250854, "sampling/importance_sampling_ratio/min": 0.3642710745334625, "sampling/importance_sampling_ratio/mean": 1.0044276714324951, "sampling/importance_sampling_ratio/max": 1.7902038097381592, "entropy": 0.560811210423708, "clip_ratio/low_mean": 0.0352126753423363, "clip_ratio/low_min": 0.0352126753423363, "clip_ratio/high_mean": 0.05272177467122674, "clip_ratio/high_max": 0.05272177467122674, "clip_ratio/region_mean": 0.08793445001356304, "reward_total_mean": 0.8054409623146057, "reward_meter_mean": 0.9542100429534912, "reward_meter_std": 0.04555569961667061, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8767144083976746, "reward_repeat_soft_std": 0.1099141389131546, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.11667262762784958, "reward_total_composite_mean": 0.8054409623146057, "reward_total_composite_std": 0.04349781945347786} {"timestamp_utc": "2026-04-13T01:46:35Z", "mode": "train", "global_step": 1386, "epoch": 0.13922651933701657, "loss": -0.1548, "grad_norm": 2.49931001663208, "learning_rate": 5.803030303030304e-06, "num_tokens": 2486185.0, "completions/mean_length": 128.0, "completions/min_length": 62.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 73.14286041259766, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.6902958750724792, "rewards/meter/std": 0.2530159056186676, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9040707349777222, "rewards/repeat_soft/std": 0.06762583553791046, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.5886595845222473, "rewards/total_composite/std": 0.26548096537590027, "reward": 0.5886595845222473, "reward_std": 0.2654809355735779, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11063520610332489, "sampling/sampling_logp_difference/max": 1.8178448677062988, "sampling/importance_sampling_ratio/min": 0.1623753160238266, "sampling/importance_sampling_ratio/mean": 1.0203309059143066, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5217544734477997, "clip_ratio/low_mean": 0.04095662012696266, "clip_ratio/low_min": 0.04095662012696266, "clip_ratio/high_mean": 0.05715795140713453, "clip_ratio/high_max": 0.05715795140713453, "clip_ratio/region_mean": 0.0981145715340972, "reward_total_mean": 0.5886595845222473, "reward_meter_mean": 0.6902958750724792, "reward_meter_std": 0.2530159056186676, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9040707349777222, "reward_repeat_soft_std": 0.06762583553791046, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.5886595845222473, "reward_total_composite_std": 0.26548096537590027} {"timestamp_utc": "2026-04-13T01:46:43Z", "mode": "train", "global_step": 1387, "epoch": 0.13932697137117026, "loss": 0.0463, "grad_norm": 6.7742414474487305, "learning_rate": 5.8e-06, "num_tokens": 2488761.0, "completions/mean_length": 119.0, "completions/min_length": 109.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.0, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.8493651151657104, "rewards/meter/std": 0.26977282762527466, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8293981552124023, "rewards/repeat_soft/std": 0.07209230959415436, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7284041047096252, "rewards/total_composite/std": 0.11751610785722733, "reward": 0.7284041047096252, "reward_std": 0.11751612275838852, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10828014463186264, "sampling/sampling_logp_difference/max": 2.4807229042053223, "sampling/importance_sampling_ratio/min": 0.08368270844221115, "sampling/importance_sampling_ratio/mean": 1.008731722831726, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7274379804730415, "clip_ratio/low_mean": 0.01720847561955452, "clip_ratio/low_min": 0.01720847561955452, "clip_ratio/high_mean": 0.08095954731106758, "clip_ratio/high_max": 0.08095954731106758, "clip_ratio/region_mean": 0.0981680229306221, "reward_total_mean": 0.7284041047096252, "reward_meter_mean": 0.8493651151657104, "reward_meter_std": 0.26977282762527466, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8293981552124023, "reward_repeat_soft_std": 0.07209230959415436, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7284041047096252, "reward_total_composite_std": 0.11751610785722733} {"timestamp_utc": "2026-04-13T01:46:51Z", "mode": "train", "global_step": 1388, "epoch": 0.13942742340532396, "loss": -0.0145, "grad_norm": 9.375272750854492, "learning_rate": 5.796969696969698e-06, "num_tokens": 2491112.0, "completions/mean_length": 103.875, "completions/min_length": 93.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.875, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.8703991770744324, "rewards/meter/std": 0.33611422777175903, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8348985910415649, "rewards/repeat_soft/std": 0.06858758628368378, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.7299820184707642, "rewards/total_composite/std": 0.14328812062740326, "reward": 0.7299820184707642, "reward_std": 0.14328812062740326, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10564558953046799, "sampling/sampling_logp_difference/max": 2.50803804397583, "sampling/importance_sampling_ratio/min": 0.08142784237861633, "sampling/importance_sampling_ratio/mean": 1.004732370376587, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6867460682988167, "clip_ratio/low_mean": 0.024327284656465054, "clip_ratio/low_min": 0.024327284656465054, "clip_ratio/high_mean": 0.07782169198617339, "clip_ratio/high_max": 0.07782169198617339, "clip_ratio/region_mean": 0.10214897664263844, "reward_total_mean": 0.7299820184707642, "reward_meter_mean": 0.8703991770744324, "reward_meter_std": 0.33611422777175903, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8348985910415649, "reward_repeat_soft_std": 0.06858758628368378, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.7299820184707642, "reward_total_composite_std": 0.14328812062740326} {"timestamp_utc": "2026-04-13T01:46:58Z", "mode": "train", "global_step": 1389, "epoch": 0.13952787543947764, "loss": -0.014, "grad_norm": 8.323498725891113, "learning_rate": 5.793939393939394e-06, "num_tokens": 2492720.0, "completions/mean_length": 51.0, "completions/min_length": 39.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.0, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.855675458908081, "rewards/meter/std": 0.29650652408599854, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.25877460837364197, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9287842512130737, "rewards/repeat_soft/std": 0.08312243223190308, "rewards/judge_quality/mean": 0.4312499761581421, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7291823625564575, "rewards/total_composite/std": 0.12727057933807373, "reward": 0.7291823625564575, "reward_std": 0.12727057933807373, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13027423620224, "sampling/sampling_logp_difference/max": 1.6543989181518555, "sampling/importance_sampling_ratio/min": 0.19120696187019348, "sampling/importance_sampling_ratio/mean": 1.0090844631195068, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8294082283973694, "clip_ratio/low_mean": 0.04216707916930318, "clip_ratio/low_min": 0.04216707916930318, "clip_ratio/high_mean": 0.07146218698471785, "clip_ratio/high_max": 0.07146218698471785, "clip_ratio/region_mean": 0.11362926615402102, "reward_total_mean": 0.7291823625564575, "reward_meter_mean": 0.855675458908081, "reward_meter_std": 0.29650652408599854, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.25877460837364197, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9287842512130737, "reward_repeat_soft_std": 0.08312243223190308, "reward_judge_quality_mean": 0.4312499761581421, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7291823625564575, "reward_total_composite_std": 0.12727057933807373} {"timestamp_utc": "2026-04-13T01:47:05Z", "mode": "train", "global_step": 1390, "epoch": 0.13962832747363135, "loss": -0.0032, "grad_norm": 9.438268661499023, "learning_rate": 5.790909090909091e-06, "num_tokens": 2494635.0, "completions/mean_length": 64.375, "completions/min_length": 58.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.375, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.732751727104187, "rewards/meter/std": 0.33979231119155884, "rewards/count_adherence/mean": 0.5, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.792303204536438, "rewards/repeat_soft/std": 0.12522943317890167, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.6287186145782471, "rewards/total_composite/std": 0.17810790240764618, "reward": 0.6287186145782471, "reward_std": 0.1781078726053238, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10077162832021713, "sampling/sampling_logp_difference/max": 3.0226759910583496, "sampling/importance_sampling_ratio/min": 0.048670802265405655, "sampling/importance_sampling_ratio/mean": 1.0109171867370605, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4853157699108124, "clip_ratio/low_mean": 0.025400209240615368, "clip_ratio/low_min": 0.025400209240615368, "clip_ratio/high_mean": 0.046238694339990616, "clip_ratio/high_max": 0.046238694339990616, "clip_ratio/region_mean": 0.07163890358060598, "reward_total_mean": 0.6287186145782471, "reward_meter_mean": 0.732751727104187, "reward_meter_std": 0.33979231119155884, "reward_count_adherence_mean": 0.5, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.792303204536438, "reward_repeat_soft_std": 0.12522943317890167, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.6287186145782471, "reward_total_composite_std": 0.17810790240764618} {"timestamp_utc": "2026-04-13T01:47:12Z", "mode": "train", "global_step": 1391, "epoch": 0.13972877950778503, "loss": 0.0837, "grad_norm": 20.780853271484375, "learning_rate": 5.787878787878788e-06, "num_tokens": 2496420.0, "completions/mean_length": 52.125, "completions/min_length": 47.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.125, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.8838033676147461, "rewards/meter/std": 0.16731953620910645, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9364285469055176, "rewards/repeat_soft/std": 0.03938061743974686, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7684794068336487, "rewards/total_composite/std": 0.07568039000034332, "reward": 0.7684794068336487, "reward_std": 0.07568038254976273, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11085618287324905, "sampling/sampling_logp_difference/max": 1.1799108982086182, "sampling/importance_sampling_ratio/min": 0.30730611085891724, "sampling/importance_sampling_ratio/mean": 1.0084482431411743, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6993229612708092, "clip_ratio/low_mean": 0.017711863853037357, "clip_ratio/low_min": 0.017711863853037357, "clip_ratio/high_mean": 0.06020479230210185, "clip_ratio/high_max": 0.06020479230210185, "clip_ratio/region_mean": 0.07791665615513921, "reward_total_mean": 0.7684794068336487, "reward_meter_mean": 0.8838033676147461, "reward_meter_std": 0.16731953620910645, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9364285469055176, "reward_repeat_soft_std": 0.03938061743974686, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7684794068336487, "reward_total_composite_std": 0.07568039000034332} {"timestamp_utc": "2026-04-13T01:47:19Z", "mode": "train", "global_step": 1392, "epoch": 0.13982923154193871, "loss": 0.001, "grad_norm": 15.973983764648438, "learning_rate": 5.784848484848486e-06, "num_tokens": 2497749.0, "completions/mean_length": 20.125, "completions/min_length": 16.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.125, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.9758648872375488, "rewards/meter/std": 0.014902104623615742, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9134190082550049, "rewards/repeat_soft/std": 0.04809791222214699, "rewards/judge_quality/mean": 0.3687500059604645, "rewards/judge_quality/std": 0.10802611708641052, "rewards/total_composite/mean": 0.791106104850769, "rewards/total_composite/std": 0.02859133668243885, "reward": 0.791106104850769, "reward_std": 0.028591321781277657, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08215148746967316, "sampling/sampling_logp_difference/max": 1.2446513175964355, "sampling/importance_sampling_ratio/min": 0.474246621131897, "sampling/importance_sampling_ratio/mean": 1.0220359563827515, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39128992334008217, "clip_ratio/low_mean": 0.03289473615586758, "clip_ratio/low_min": 0.03289473615586758, "clip_ratio/high_mean": 0.04556728061288595, "clip_ratio/high_max": 0.04556728061288595, "clip_ratio/region_mean": 0.07846201676875353, "reward_total_mean": 0.791106104850769, "reward_meter_mean": 0.9758648872375488, "reward_meter_std": 0.014902104623615742, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9134190082550049, "reward_repeat_soft_std": 0.04809791222214699, "reward_judge_quality_mean": 0.3687500059604645, "reward_judge_quality_std": 0.10802611708641052, "reward_total_composite_mean": 0.791106104850769, "reward_total_composite_std": 0.02859133668243885} {"timestamp_utc": "2026-04-13T01:47:26Z", "mode": "train", "global_step": 1393, "epoch": 0.13992968357609242, "loss": -0.0, "grad_norm": 16.507516860961914, "learning_rate": 5.781818181818181e-06, "num_tokens": 2499054.0, "completions/mean_length": 19.125, "completions/min_length": 16.0, "completions/max_length": 21.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.125, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 21.0, "rewards/meter/mean": 0.9892181158065796, "rewards/meter/std": 0.004955657757818699, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9141532778739929, "rewards/repeat_soft/std": 0.09274622052907944, "rewards/judge_quality/mean": 0.36249998211860657, "rewards/judge_quality/std": 0.12464234232902527, "rewards/total_composite/mean": 0.7953134775161743, "rewards/total_composite/std": 0.042840730398893356, "reward": 0.7953134775161743, "reward_std": 0.042840734124183655, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11558478325605392, "sampling/sampling_logp_difference/max": 2.613626718521118, "sampling/importance_sampling_ratio/min": 0.0732683390378952, "sampling/importance_sampling_ratio/mean": 0.9869768023490906, "sampling/importance_sampling_ratio/max": 1.5706106424331665, "entropy": 0.47026442736387253, "clip_ratio/low_mean": 0.04852204071357846, "clip_ratio/low_min": 0.04852204071357846, "clip_ratio/high_mean": 0.09322479134425521, "clip_ratio/high_max": 0.09322479134425521, "clip_ratio/region_mean": 0.14174683205783367, "reward_total_mean": 0.7953134775161743, "reward_meter_mean": 0.9892181158065796, "reward_meter_std": 0.004955657757818699, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9141532778739929, "reward_repeat_soft_std": 0.09274622052907944, "reward_judge_quality_mean": 0.36249998211860657, "reward_judge_quality_std": 0.12464234232902527, "reward_total_composite_mean": 0.7953134775161743, "reward_total_composite_std": 0.042840730398893356} {"timestamp_utc": "2026-04-13T01:47:37Z", "mode": "train", "global_step": 1394, "epoch": 0.1400301356102461, "loss": 0.644, "grad_norm": 4.102493762969971, "learning_rate": 5.7787878787878795e-06, "num_tokens": 2501751.0, "completions/mean_length": 177.125, "completions/min_length": 59.0, "completions/max_length": 481.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 177.125, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 481.0, "rewards/meter/mean": 0.9914822578430176, "rewards/meter/std": 0.011052754707634449, "rewards/count_adherence/mean": 0.25, "rewards/count_adherence/std": 0.26726123690605164, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6673406362533569, "rewards/repeat_soft/std": 0.15230603516101837, "rewards/judge_quality/mean": 0.3387500047683716, "rewards/judge_quality/std": 0.09538455307483673, "rewards/total_composite/mean": 0.6520260572433472, "rewards/total_composite/std": 0.06881435215473175, "reward": 0.6520260572433472, "reward_std": 0.06881435960531235, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06901873648166656, "sampling/sampling_logp_difference/max": 2.032135009765625, "sampling/importance_sampling_ratio/min": 0.13105541467666626, "sampling/importance_sampling_ratio/mean": 1.009684443473816, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6640314050018787, "clip_ratio/low_mean": 0.02709387382492423, "clip_ratio/low_min": 0.02709387382492423, "clip_ratio/high_mean": 0.06047848518937826, "clip_ratio/high_max": 0.06047848518937826, "clip_ratio/region_mean": 0.08757235901430249, "reward_total_mean": 0.6520260572433472, "reward_meter_mean": 0.9914822578430176, "reward_meter_std": 0.011052754707634449, "reward_count_adherence_mean": 0.25, "reward_count_adherence_std": 0.26726123690605164, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6673406362533569, "reward_repeat_soft_std": 0.15230603516101837, "reward_judge_quality_mean": 0.3387500047683716, "reward_judge_quality_std": 0.09538455307483673, "reward_total_composite_mean": 0.6520260572433472, "reward_total_composite_std": 0.06881435215473175} {"timestamp_utc": "2026-04-13T01:47:44Z", "mode": "train", "global_step": 1395, "epoch": 0.1401305876443998, "loss": -0.0084, "grad_norm": 9.659751892089844, "learning_rate": 5.775757575757577e-06, "num_tokens": 2503261.0, "completions/mean_length": 43.75, "completions/min_length": 39.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.75, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.8822606801986694, "rewards/meter/std": 0.22000053524971008, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9631691575050354, "rewards/repeat_soft/std": 0.025440096855163574, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.8255842328071594, "rewards/total_composite/std": 0.09850581735372543, "reward": 0.8255842328071594, "reward_std": 0.09850580245256424, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09699435532093048, "sampling/sampling_logp_difference/max": 1.2006406784057617, "sampling/importance_sampling_ratio/min": 0.30100131034851074, "sampling/importance_sampling_ratio/mean": 0.9982885718345642, "sampling/importance_sampling_ratio/max": 1.6401164531707764, "entropy": 0.629560973495245, "clip_ratio/low_mean": 0.059066420421004295, "clip_ratio/low_min": 0.059066420421004295, "clip_ratio/high_mean": 0.029880952090024948, "clip_ratio/high_max": 0.029880952090024948, "clip_ratio/region_mean": 0.08894737251102924, "reward_total_mean": 0.8255842328071594, "reward_meter_mean": 0.8822606801986694, "reward_meter_std": 0.22000053524971008, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9631691575050354, "reward_repeat_soft_std": 0.025440096855163574, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.8255842328071594, "reward_total_composite_std": 0.09850581735372543} {"timestamp_utc": "2026-04-13T01:47:51Z", "mode": "train", "global_step": 1396, "epoch": 0.1402310396785535, "loss": 0.0694, "grad_norm": 11.951143264770508, "learning_rate": 5.772727272727273e-06, "num_tokens": 2504614.0, "completions/mean_length": 26.125, "completions/min_length": 19.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.125, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9872394800186157, "rewards/meter/std": 0.009352468885481358, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9244725704193115, "rewards/repeat_soft/std": 0.058005962520837784, "rewards/judge_quality/mean": 0.4087499976158142, "rewards/judge_quality/std": 0.10507649928331375, "rewards/total_composite/mean": 0.8093299865722656, "rewards/total_composite/std": 0.03548453748226166, "reward": 0.8093299865722656, "reward_std": 0.035484541207551956, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1542372703552246, "sampling/sampling_logp_difference/max": 1.286785364151001, "sampling/importance_sampling_ratio/min": 0.27615711092948914, "sampling/importance_sampling_ratio/mean": 1.035102367401123, "sampling/importance_sampling_ratio/max": 1.9547375440597534, "entropy": 1.3327905610203743, "clip_ratio/low_mean": 0.01293103490024805, "clip_ratio/low_min": 0.01293103490024805, "clip_ratio/high_mean": 0.10382317751646042, "clip_ratio/high_max": 0.10382317751646042, "clip_ratio/region_mean": 0.11675421241670847, "reward_total_mean": 0.8093299865722656, "reward_meter_mean": 0.9872394800186157, "reward_meter_std": 0.009352468885481358, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9244725704193115, "reward_repeat_soft_std": 0.058005962520837784, "reward_judge_quality_mean": 0.4087499976158142, "reward_judge_quality_std": 0.10507649928331375, "reward_total_composite_mean": 0.8093299865722656, "reward_total_composite_std": 0.03548453748226166} {"timestamp_utc": "2026-04-13T01:47:58Z", "mode": "train", "global_step": 1397, "epoch": 0.14033149171270717, "loss": 0.0282, "grad_norm": 9.627054214477539, "learning_rate": 5.76969696969697e-06, "num_tokens": 2506283.0, "completions/mean_length": 43.625, "completions/min_length": 37.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.625, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9795829057693481, "rewards/meter/std": 0.025600772351026535, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8601875305175781, "rewards/repeat_soft/std": 0.13404761254787445, "rewards/judge_quality/mean": 0.3474999964237213, "rewards/judge_quality/std": 0.11310551315546036, "rewards/total_composite/mean": 0.7810810804367065, "rewards/total_composite/std": 0.039621949195861816, "reward": 0.7810810804367065, "reward_std": 0.03962194547057152, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11368395388126373, "sampling/sampling_logp_difference/max": 1.2293825149536133, "sampling/importance_sampling_ratio/min": 0.29247310757637024, "sampling/importance_sampling_ratio/mean": 1.014564871788025, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7785454019904137, "clip_ratio/low_mean": 0.03137112455442548, "clip_ratio/low_min": 0.03137112455442548, "clip_ratio/high_mean": 0.06782927177846432, "clip_ratio/high_max": 0.06782927177846432, "clip_ratio/region_mean": 0.0992003963328898, "reward_total_mean": 0.7810810804367065, "reward_meter_mean": 0.9795829057693481, "reward_meter_std": 0.025600772351026535, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8601875305175781, "reward_repeat_soft_std": 0.13404761254787445, "reward_judge_quality_mean": 0.3474999964237213, "reward_judge_quality_std": 0.11310551315546036, "reward_total_composite_mean": 0.7810810804367065, "reward_total_composite_std": 0.039621949195861816} {"timestamp_utc": "2026-04-13T01:48:05Z", "mode": "train", "global_step": 1398, "epoch": 0.14043194374686088, "loss": 0.1053, "grad_norm": 12.771595001220703, "learning_rate": 5.766666666666667e-06, "num_tokens": 2508076.0, "completions/mean_length": 56.125, "completions/min_length": 42.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.125, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.7846017479896545, "rewards/meter/std": 0.26551300287246704, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.856947124004364, "rewards/repeat_soft/std": 0.11223728954792023, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.6596404910087585, "rewards/total_composite/std": 0.1337212175130844, "reward": 0.6596404910087585, "reward_std": 0.1337212175130844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15407182276248932, "sampling/sampling_logp_difference/max": 5.059327125549316, "sampling/importance_sampling_ratio/min": 0.006349830888211727, "sampling/importance_sampling_ratio/mean": 1.0117796659469604, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1077325493097305, "clip_ratio/low_mean": 0.03205121261999011, "clip_ratio/low_min": 0.03205121261999011, "clip_ratio/high_mean": 0.08255061320960522, "clip_ratio/high_max": 0.08255061320960522, "clip_ratio/region_mean": 0.11460182582959533, "reward_total_mean": 0.6596404910087585, "reward_meter_mean": 0.7846017479896545, "reward_meter_std": 0.26551300287246704, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.856947124004364, "reward_repeat_soft_std": 0.11223728954792023, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.6596404910087585, "reward_total_composite_std": 0.1337212175130844} {"timestamp_utc": "2026-04-13T01:48:13Z", "mode": "train", "global_step": 1399, "epoch": 0.14053239578101456, "loss": 0.0489, "grad_norm": 10.797470092773438, "learning_rate": 5.763636363636365e-06, "num_tokens": 2509797.0, "completions/mean_length": 57.125, "completions/min_length": 44.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.125, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.8951253890991211, "rewards/meter/std": 0.2560117244720459, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.25877460837364197, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.949734628200531, "rewards/repeat_soft/std": 0.04145578667521477, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.19334925711154938, "rewards/total_composite/mean": 0.7602798938751221, "rewards/total_composite/std": 0.13365118205547333, "reward": 0.7602798938751221, "reward_std": 0.13365118205547333, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12116154283285141, "sampling/sampling_logp_difference/max": 2.274878978729248, "sampling/importance_sampling_ratio/min": 0.3045061230659485, "sampling/importance_sampling_ratio/mean": 1.0229878425598145, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8901132494211197, "clip_ratio/low_mean": 0.05604823585599661, "clip_ratio/low_min": 0.05604823585599661, "clip_ratio/high_mean": 0.04954442009329796, "clip_ratio/high_max": 0.04954442009329796, "clip_ratio/region_mean": 0.10559265594929457, "reward_total_mean": 0.7602798938751221, "reward_meter_mean": 0.8951253890991211, "reward_meter_std": 0.2560117244720459, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.25877460837364197, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.949734628200531, "reward_repeat_soft_std": 0.04145578667521477, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.19334925711154938, "reward_total_composite_mean": 0.7602798938751221, "reward_total_composite_std": 0.13365118205547333} {"timestamp_utc": "2026-04-13T01:48:24Z", "mode": "train", "global_step": 1400, "epoch": 0.14063284781516824, "loss": -0.1305, "grad_norm": 3.378298282623291, "learning_rate": 5.760606060606061e-06, "num_tokens": 2511652.0, "completions/mean_length": 123.875, "completions/min_length": 66.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 68.42857360839844, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.6510429382324219, "rewards/meter/std": 0.4372636079788208, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8919229507446289, "rewards/repeat_soft/std": 0.06088212877511978, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.5996390581130981, "rewards/total_composite/std": 0.30529698729515076, "reward": 0.5996390581130981, "reward_std": 0.30529701709747314, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12195756286382675, "sampling/sampling_logp_difference/max": 1.2129743099212646, "sampling/importance_sampling_ratio/min": 0.2973116636276245, "sampling/importance_sampling_ratio/mean": 1.0219744443893433, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8771016448736191, "clip_ratio/low_mean": 0.0183322518132627, "clip_ratio/low_min": 0.0183322518132627, "clip_ratio/high_mean": 0.06914506200700998, "clip_ratio/high_max": 0.06914506200700998, "clip_ratio/region_mean": 0.08747731382027268, "reward_total_mean": 0.5996390581130981, "reward_meter_mean": 0.6510429382324219, "reward_meter_std": 0.4372636079788208, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8919229507446289, "reward_repeat_soft_std": 0.06088212877511978, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.5996390581130981, "reward_total_composite_std": 0.30529698729515076} {"timestamp_utc": "2026-04-13T01:49:22Z", "mode": "eval", "global_step": 1400, "epoch": 0.14063284781516824, "eval_loss": NaN, "eval_runtime": 57.3302, "eval_samples_per_second": 1.395, "eval_steps_per_second": 0.174, "eval_num_tokens": 2511652.0, "eval_completions/mean_length": 102.55, "eval_completions/min_length": 36.6, "eval_completions/max_length": 238.9, "eval_completions/clipped_ratio": 0.0625, "eval_completions/mean_terminated_length": 75.85464401245117, "eval_completions/min_terminated_length": 36.6, "eval_completions/max_terminated_length": 122.6, "eval_rewards/meter/mean": 0.7912608325481415, "eval_rewards/meter/std": 0.3008030079305172, "eval_rewards/count_adherence/mean": 0.9297916650772095, "eval_rewards/count_adherence/std": 0.16473203301429748, "eval_rewards/hard_gate/mean": 0.9375, "eval_rewards/hard_gate/std": 0.12246559858322144, "eval_rewards/repeat_soft/mean": 0.8626482009887695, "eval_rewards/repeat_soft/std": 0.11765107586979866, "eval_rewards/judge_quality/mean": 0.3598750025033951, "eval_rewards/judge_quality/std": 0.12206062600016594, "eval_rewards/total_composite/mean": 0.6679169416427613, "eval_rewards/total_composite/std": 0.18861824981868267, "eval_reward": 0.6679169416427613, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.06885084137320518, "eval_sampling/sampling_logp_difference/max": 0.9467163562774659, "eval_sampling/importance_sampling_ratio/min": 0.3914416402578354, "eval_sampling/importance_sampling_ratio/mean": 1.0177271366119385, "eval_sampling/importance_sampling_ratio/max": 1.509192740917206, "eval_entropy": 0.7616556942462921, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6679169416427613, "eval_reward_meter_mean": 0.7912608325481415, "eval_reward_meter_std": 0.3008030079305172, "eval_reward_count_adherence_mean": 0.9297916650772095, "eval_reward_count_adherence_std": 0.16473203301429748, "eval_reward_hard_gate_mean": 0.9375, "eval_reward_hard_gate_std": 0.12246559858322144, "eval_reward_repeat_soft_mean": 0.8626482009887695, "eval_reward_repeat_soft_std": 0.11765107586979866, "eval_reward_judge_quality_mean": 0.3598750025033951, "eval_reward_judge_quality_std": 0.12206062600016594, "eval_reward_total_composite_mean": 0.6679169416427613, "eval_reward_total_composite_std": 0.18861824981868267} {"timestamp_utc": "2026-04-13T01:49:38Z", "mode": "train", "global_step": 1401, "epoch": 0.14073329984932195, "loss": -0.1018, "grad_norm": 3.2747840881347656, "learning_rate": 5.7575757575757586e-06, "num_tokens": 2512984.0, "completions/mean_length": 95.5, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 36.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.5369347333908081, "rewards/meter/std": 0.3698643743991852, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.979684591293335, "rewards/repeat_soft/std": 0.022118287160992622, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.25560852885246277, "rewards/total_composite/mean": 0.577632486820221, "rewards/total_composite/std": 0.2650573253631592, "reward": 0.577632486820221, "reward_std": 0.2650573253631592, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1813562661409378, "sampling/sampling_logp_difference/max": 1.5798922777175903, "sampling/importance_sampling_ratio/min": 0.2059972733259201, "sampling/importance_sampling_ratio/mean": 1.0446075201034546, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.217515766620636, "clip_ratio/low_mean": 0.042031003162264824, "clip_ratio/low_min": 0.042031003162264824, "clip_ratio/high_mean": 0.08596994820982218, "clip_ratio/high_max": 0.08596994820982218, "clip_ratio/region_mean": 0.128000951372087, "reward_total_mean": 0.577632486820221, "reward_meter_mean": 0.5369347333908081, "reward_meter_std": 0.3698643743991852, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.979684591293335, "reward_repeat_soft_std": 0.022118287160992622, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.25560852885246277, "reward_total_composite_mean": 0.577632486820221, "reward_total_composite_std": 0.2650573253631592} {"timestamp_utc": "2026-04-13T01:49:51Z", "mode": "train", "global_step": 1402, "epoch": 0.14083375188347563, "loss": -0.1486, "grad_norm": 1.9033310413360596, "learning_rate": 5.754545454545455e-06, "num_tokens": 2514648.0, "completions/mean_length": 179.0, "completions/min_length": 62.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.6753156185150146, "rewards/meter/std": 0.44805100560188293, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9639501571655273, "rewards/repeat_soft/std": 0.0397409088909626, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.2539263069629669, "rewards/total_composite/mean": 0.6029170155525208, "rewards/total_composite/std": 0.3859651982784271, "reward": 0.6029170155525208, "reward_std": 0.3859651982784271, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1361982673406601, "sampling/sampling_logp_difference/max": 1.8610765933990479, "sampling/importance_sampling_ratio/min": 0.15550512075424194, "sampling/importance_sampling_ratio/mean": 1.0149407386779785, "sampling/importance_sampling_ratio/max": 1.8542344570159912, "entropy": 0.7548442855477333, "clip_ratio/low_mean": 0.014925372786819935, "clip_ratio/low_min": 0.014925372786819935, "clip_ratio/high_mean": 0.07126723974943161, "clip_ratio/high_max": 0.07126723974943161, "clip_ratio/region_mean": 0.08619261253625154, "reward_total_mean": 0.6029170155525208, "reward_meter_mean": 0.6753156185150146, "reward_meter_std": 0.44805100560188293, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9639501571655273, "reward_repeat_soft_std": 0.0397409088909626, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.2539263069629669, "reward_total_composite_mean": 0.6029170155525208, "reward_total_composite_std": 0.3859651982784271} {"timestamp_utc": "2026-04-13T01:50:04Z", "mode": "train", "global_step": 1403, "epoch": 0.14093420391762934, "loss": -0.0968, "grad_norm": 2.8512179851531982, "learning_rate": 5.751515151515152e-06, "num_tokens": 2516370.0, "completions/mean_length": 99.25, "completions/min_length": 38.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 40.28571701049805, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.47347259521484375, "rewards/meter/std": 0.4035593569278717, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.955089807510376, "rewards/repeat_soft/std": 0.02978280372917652, "rewards/judge_quality/mean": 0.4050000011920929, "rewards/judge_quality/std": 0.2523036599159241, "rewards/total_composite/mean": 0.5426303148269653, "rewards/total_composite/std": 0.27030622959136963, "reward": 0.5426303148269653, "reward_std": 0.27030622959136963, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10878945142030716, "sampling/sampling_logp_difference/max": 1.077998161315918, "sampling/importance_sampling_ratio/min": 0.3402760326862335, "sampling/importance_sampling_ratio/mean": 1.0121022462844849, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7730866372585297, "clip_ratio/low_mean": 0.0343750009778887, "clip_ratio/low_min": 0.0343750009778887, "clip_ratio/high_mean": 0.042822967283427715, "clip_ratio/high_max": 0.042822967283427715, "clip_ratio/region_mean": 0.07719796826131642, "reward_total_mean": 0.5426303148269653, "reward_meter_mean": 0.47347259521484375, "reward_meter_std": 0.4035593569278717, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.955089807510376, "reward_repeat_soft_std": 0.02978280372917652, "reward_judge_quality_mean": 0.4050000011920929, "reward_judge_quality_std": 0.2523036599159241, "reward_total_composite_mean": 0.5426303148269653, "reward_total_composite_std": 0.27030622959136963} {"timestamp_utc": "2026-04-13T01:50:12Z", "mode": "train", "global_step": 1404, "epoch": 0.14103465595178302, "loss": 0.0526, "grad_norm": 9.241071701049805, "learning_rate": 5.748484848484849e-06, "num_tokens": 2518714.0, "completions/mean_length": 93.0, "completions/min_length": 81.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.0, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.6513051986694336, "rewards/meter/std": 0.27180084586143494, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8885359764099121, "rewards/repeat_soft/std": 0.08612842857837677, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.1249857097864151, "rewards/total_composite/mean": 0.6330034732818604, "rewards/total_composite/std": 0.1362035721540451, "reward": 0.6330034732818604, "reward_std": 0.1362035721540451, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11955065280199051, "sampling/sampling_logp_difference/max": 1.465968132019043, "sampling/importance_sampling_ratio/min": 0.23085439205169678, "sampling/importance_sampling_ratio/mean": 1.0168516635894775, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.948821671307087, "clip_ratio/low_mean": 0.050865483935922384, "clip_ratio/low_min": 0.050865483935922384, "clip_ratio/high_mean": 0.028958806302398443, "clip_ratio/high_max": 0.028958806302398443, "clip_ratio/region_mean": 0.07982429023832083, "reward_total_mean": 0.6330034732818604, "reward_meter_mean": 0.6513051986694336, "reward_meter_std": 0.27180084586143494, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8885359764099121, "reward_repeat_soft_std": 0.08612842857837677, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.1249857097864151, "reward_total_composite_mean": 0.6330034732818604, "reward_total_composite_std": 0.1362035721540451} {"timestamp_utc": "2026-04-13T01:50:18Z", "mode": "train", "global_step": 1405, "epoch": 0.1411351079859367, "loss": -0.0009, "grad_norm": 9.862833023071289, "learning_rate": 5.745454545454546e-06, "num_tokens": 2520385.0, "completions/mean_length": 51.875, "completions/min_length": 47.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.875, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.987746000289917, "rewards/meter/std": 0.021429013460874557, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8815933465957642, "rewards/repeat_soft/std": 0.09424660354852676, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.8210200071334839, "rewards/total_composite/std": 0.03461642563343048, "reward": 0.8210200071334839, "reward_std": 0.03461640328168869, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12291567027568817, "sampling/sampling_logp_difference/max": 2.2882673740386963, "sampling/importance_sampling_ratio/min": 0.20217101275920868, "sampling/importance_sampling_ratio/mean": 1.0218040943145752, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9215101301670074, "clip_ratio/low_mean": 0.08908182289451361, "clip_ratio/low_min": 0.08908182289451361, "clip_ratio/high_mean": 0.0759897418320179, "clip_ratio/high_max": 0.0759897418320179, "clip_ratio/region_mean": 0.1650715647265315, "reward_total_mean": 0.8210200071334839, "reward_meter_mean": 0.987746000289917, "reward_meter_std": 0.021429013460874557, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8815933465957642, "reward_repeat_soft_std": 0.09424660354852676, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.8210200071334839, "reward_total_composite_std": 0.03461642563343048} {"timestamp_utc": "2026-04-13T01:50:30Z", "mode": "train", "global_step": 1406, "epoch": 0.1412355600200904, "loss": -0.1673, "grad_norm": 1.8474845886230469, "learning_rate": 5.742424242424242e-06, "num_tokens": 2522267.0, "completions/mean_length": 124.25, "completions/min_length": 62.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 68.85714721679688, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.8780786991119385, "rewards/meter/std": 0.2165718823671341, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8586013317108154, "rewards/repeat_soft/std": 0.09256032109260559, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.6847710609436035, "rewards/total_composite/std": 0.2779462933540344, "reward": 0.6847710609436035, "reward_std": 0.27794626355171204, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09560170024633408, "sampling/sampling_logp_difference/max": 1.9316949844360352, "sampling/importance_sampling_ratio/min": 0.14490237832069397, "sampling/importance_sampling_ratio/mean": 1.01205313205719, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5072112046182156, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08224107255227864, "clip_ratio/high_max": 0.08224107255227864, "clip_ratio/region_mean": 0.08224107255227864, "reward_total_mean": 0.6847710609436035, "reward_meter_mean": 0.8780786991119385, "reward_meter_std": 0.2165718823671341, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.2651650309562683, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8586013317108154, "reward_repeat_soft_std": 0.09256032109260559, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.6847710609436035, "reward_total_composite_std": 0.2779462933540344} {"timestamp_utc": "2026-04-13T01:50:36Z", "mode": "train", "global_step": 1407, "epoch": 0.1413360120542441, "loss": -0.0036, "grad_norm": 19.869630813598633, "learning_rate": 5.73939393939394e-06, "num_tokens": 2523675.0, "completions/mean_length": 21.0, "completions/min_length": 17.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.0, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.8301438093185425, "rewards/meter/std": 0.22857604920864105, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9020711183547974, "rewards/repeat_soft/std": 0.08975186944007874, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.2562922537326813, "rewards/total_composite/mean": 0.7292718291282654, "rewards/total_composite/std": 0.12849077582359314, "reward": 0.7292718291282654, "reward_std": 0.12849077582359314, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14348740875720978, "sampling/sampling_logp_difference/max": 2.1666100025177, "sampling/importance_sampling_ratio/min": 0.11456532776355743, "sampling/importance_sampling_ratio/mean": 1.0149476528167725, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.024291031062603, "clip_ratio/low_mean": 0.06414496712386608, "clip_ratio/low_min": 0.06414496712386608, "clip_ratio/high_mean": 0.06345991138368845, "clip_ratio/high_max": 0.06345991138368845, "clip_ratio/region_mean": 0.12760487850755453, "reward_total_mean": 0.7292718291282654, "reward_meter_mean": 0.8301438093185425, "reward_meter_std": 0.22857604920864105, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9020711183547974, "reward_repeat_soft_std": 0.08975186944007874, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.2562922537326813, "reward_total_composite_mean": 0.7292718291282654, "reward_total_composite_std": 0.12849077582359314} {"timestamp_utc": "2026-04-13T01:50:43Z", "mode": "train", "global_step": 1408, "epoch": 0.1414364640883978, "loss": -0.0007, "grad_norm": 12.216288566589355, "learning_rate": 5.736363636363637e-06, "num_tokens": 2525170.0, "completions/mean_length": 30.875, "completions/min_length": 30.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.875, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.7979118824005127, "rewards/meter/std": 0.22237098217010498, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8021788597106934, "rewards/repeat_soft/std": 0.0575748048722744, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.10260014235973358, "rewards/total_composite/mean": 0.7299032211303711, "rewards/total_composite/std": 0.10978297889232635, "reward": 0.7299032211303711, "reward_std": 0.10978297889232635, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07083958387374878, "sampling/sampling_logp_difference/max": 1.326988697052002, "sampling/importance_sampling_ratio/min": 0.2769761085510254, "sampling/importance_sampling_ratio/mean": 1.0233995914459229, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2847901303321123, "clip_ratio/low_mean": 0.04153225850313902, "clip_ratio/low_min": 0.04153225850313902, "clip_ratio/high_mean": 0.01576704578474164, "clip_ratio/high_max": 0.01576704578474164, "clip_ratio/region_mean": 0.05729930428788066, "reward_total_mean": 0.7299032211303711, "reward_meter_mean": 0.7979118824005127, "reward_meter_std": 0.22237098217010498, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8021788597106934, "reward_repeat_soft_std": 0.0575748048722744, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.10260014235973358, "reward_total_composite_mean": 0.7299032211303711, "reward_total_composite_std": 0.10978297889232635} {"timestamp_utc": "2026-04-13T01:50:50Z", "mode": "train", "global_step": 1409, "epoch": 0.14153691612255148, "loss": 0.0049, "grad_norm": 7.10624885559082, "learning_rate": 5.733333333333334e-06, "num_tokens": 2527366.0, "completions/mean_length": 96.5, "completions/min_length": 73.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.5, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.5368396043777466, "rewards/meter/std": 0.3423577845096588, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8176373839378357, "rewards/repeat_soft/std": 0.10198806971311569, "rewards/judge_quality/mean": 0.3137499690055847, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.5599665641784668, "rewards/total_composite/std": 0.14019866287708282, "reward": 0.5599665641784668, "reward_std": 0.14019863307476044, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12359217554330826, "sampling/sampling_logp_difference/max": 1.7214627265930176, "sampling/importance_sampling_ratio/min": 0.178804412484169, "sampling/importance_sampling_ratio/mean": 1.0174647569656372, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.022881768643856, "clip_ratio/low_mean": 0.03427149914205074, "clip_ratio/low_min": 0.03427149914205074, "clip_ratio/high_mean": 0.06646061781793833, "clip_ratio/high_max": 0.06646061781793833, "clip_ratio/region_mean": 0.10073211695998907, "reward_total_mean": 0.5599665641784668, "reward_meter_mean": 0.5368396043777466, "reward_meter_std": 0.3423577845096588, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8176373839378357, "reward_repeat_soft_std": 0.10198806971311569, "reward_judge_quality_mean": 0.3137499690055847, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.5599665641784668, "reward_total_composite_std": 0.14019866287708282} {"timestamp_utc": "2026-04-13T01:50:58Z", "mode": "train", "global_step": 1410, "epoch": 0.14163736815670516, "loss": 0.0143, "grad_norm": 9.467467308044434, "learning_rate": 5.7303030303030305e-06, "num_tokens": 2529096.0, "completions/mean_length": 48.25, "completions/min_length": 42.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.25, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.700434148311615, "rewards/meter/std": 0.39340269565582275, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8713728785514832, "rewards/repeat_soft/std": 0.09002815186977386, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.6895827054977417, "rewards/total_composite/std": 0.1880435347557068, "reward": 0.6895827054977417, "reward_std": 0.1880435347557068, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10656866431236267, "sampling/sampling_logp_difference/max": 1.9078621864318848, "sampling/importance_sampling_ratio/min": 0.148397296667099, "sampling/importance_sampling_ratio/mean": 1.003231406211853, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.745173953473568, "clip_ratio/low_mean": 0.031239313073456287, "clip_ratio/low_min": 0.031239313073456287, "clip_ratio/high_mean": 0.0837126150727272, "clip_ratio/high_max": 0.0837126150727272, "clip_ratio/region_mean": 0.11495192814618349, "reward_total_mean": 0.6895827054977417, "reward_meter_mean": 0.700434148311615, "reward_meter_std": 0.39340269565582275, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8713728785514832, "reward_repeat_soft_std": 0.09002815186977386, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.6895827054977417, "reward_total_composite_std": 0.1880435347557068} {"timestamp_utc": "2026-04-13T01:51:04Z", "mode": "train", "global_step": 1411, "epoch": 0.14173782019085887, "loss": 0.0, "grad_norm": 9.728071212768555, "learning_rate": 5.727272727272728e-06, "num_tokens": 2530781.0, "completions/mean_length": 40.625, "completions/min_length": 33.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.625, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9047913551330566, "rewards/meter/std": 0.21505241096019745, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.746515154838562, "rewards/repeat_soft/std": 0.12399885058403015, "rewards/judge_quality/mean": 0.35999998450279236, "rewards/judge_quality/std": 0.09165150672197342, "rewards/total_composite/mean": 0.7398076057434082, "rewards/total_composite/std": 0.08831783384084702, "reward": 0.7398076057434082, "reward_std": 0.08831784129142761, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1002759262919426, "sampling/sampling_logp_difference/max": 1.6052831411361694, "sampling/importance_sampling_ratio/min": 0.20083267986774445, "sampling/importance_sampling_ratio/mean": 1.0032875537872314, "sampling/importance_sampling_ratio/max": 1.7270715236663818, "entropy": 0.6492675691843033, "clip_ratio/low_mean": 0.049454087391495705, "clip_ratio/low_min": 0.049454087391495705, "clip_ratio/high_mean": 0.04629104398190975, "clip_ratio/high_max": 0.04629104398190975, "clip_ratio/region_mean": 0.09574513137340546, "reward_total_mean": 0.7398076057434082, "reward_meter_mean": 0.9047913551330566, "reward_meter_std": 0.21505241096019745, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.746515154838562, "reward_repeat_soft_std": 0.12399885058403015, "reward_judge_quality_mean": 0.35999998450279236, "reward_judge_quality_std": 0.09165150672197342, "reward_total_composite_mean": 0.7398076057434082, "reward_total_composite_std": 0.08831783384084702} {"timestamp_utc": "2026-04-13T01:51:12Z", "mode": "train", "global_step": 1412, "epoch": 0.14183827222501255, "loss": 0.0289, "grad_norm": 6.870090484619141, "learning_rate": 5.724242424242424e-06, "num_tokens": 2532997.0, "completions/mean_length": 104.0, "completions/min_length": 99.0, "completions/max_length": 113.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.0, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.9951931238174438, "rewards/meter/std": 0.0032116337679326534, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8075200319290161, "rewards/repeat_soft/std": 0.09991387277841568, "rewards/judge_quality/mean": 0.30124998092651367, "rewards/judge_quality/std": 0.10398317128419876, "rewards/total_composite/mean": 0.7689639329910278, "rewards/total_composite/std": 0.033206045627593994, "reward": 0.7689639329910278, "reward_std": 0.033206041902303696, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11549609899520874, "sampling/sampling_logp_difference/max": 2.414828300476074, "sampling/importance_sampling_ratio/min": 0.08938268572092056, "sampling/importance_sampling_ratio/mean": 1.0136935710906982, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0406280606985092, "clip_ratio/low_mean": 0.05680523533374071, "clip_ratio/low_min": 0.05680523533374071, "clip_ratio/high_mean": 0.04330406058579683, "clip_ratio/high_max": 0.04330406058579683, "clip_ratio/region_mean": 0.10010929591953754, "reward_total_mean": 0.7689639329910278, "reward_meter_mean": 0.9951931238174438, "reward_meter_std": 0.0032116337679326534, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8075200319290161, "reward_repeat_soft_std": 0.09991387277841568, "reward_judge_quality_mean": 0.30124998092651367, "reward_judge_quality_std": 0.10398317128419876, "reward_total_composite_mean": 0.7689639329910278, "reward_total_composite_std": 0.033206045627593994} {"timestamp_utc": "2026-04-13T01:51:20Z", "mode": "train", "global_step": 1413, "epoch": 0.14193872425916626, "loss": 0.1498, "grad_norm": 9.73365306854248, "learning_rate": 5.721212121212122e-06, "num_tokens": 2535006.0, "completions/mean_length": 76.125, "completions/min_length": 59.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.125, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.9923202991485596, "rewards/meter/std": 0.003936809487640858, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9012033939361572, "rewards/repeat_soft/std": 0.0457603894174099, "rewards/judge_quality/mean": 0.48125001788139343, "rewards/judge_quality/std": 0.2820049524307251, "rewards/total_composite/mean": 0.8185394406318665, "rewards/total_composite/std": 0.10305140167474747, "reward": 0.8185394406318665, "reward_std": 0.10305140167474747, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16770143806934357, "sampling/sampling_logp_difference/max": 1.922398567199707, "sampling/importance_sampling_ratio/min": 0.1462557464838028, "sampling/importance_sampling_ratio/mean": 1.0205634832382202, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3645390197634697, "clip_ratio/low_mean": 0.0970873860642314, "clip_ratio/low_min": 0.0970873860642314, "clip_ratio/high_mean": 0.03455159813165665, "clip_ratio/high_max": 0.03455159813165665, "clip_ratio/region_mean": 0.13163898419588804, "reward_total_mean": 0.8185394406318665, "reward_meter_mean": 0.9923202991485596, "reward_meter_std": 0.003936809487640858, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9012033939361572, "reward_repeat_soft_std": 0.0457603894174099, "reward_judge_quality_mean": 0.48125001788139343, "reward_judge_quality_std": 0.2820049524307251, "reward_total_composite_mean": 0.8185394406318665, "reward_total_composite_std": 0.10305140167474747} {"timestamp_utc": "2026-04-13T01:51:27Z", "mode": "train", "global_step": 1414, "epoch": 0.14203917629331994, "loss": 0.0474, "grad_norm": 11.16933822631836, "learning_rate": 5.718181818181819e-06, "num_tokens": 2536531.0, "completions/mean_length": 43.625, "completions/min_length": 40.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.625, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9893183708190918, "rewards/meter/std": 0.011413329280912876, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9586776494979858, "rewards/repeat_soft/std": 0.03852525353431702, "rewards/judge_quality/mean": 0.5887500047683716, "rewards/judge_quality/std": 0.21596544981002808, "rewards/total_composite/mean": 0.8676860332489014, "rewards/total_composite/std": 0.06873012334108353, "reward": 0.8676860332489014, "reward_std": 0.06873010843992233, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.128998264670372, "sampling/sampling_logp_difference/max": 1.1419916152954102, "sampling/importance_sampling_ratio/min": 0.31918269395828247, "sampling/importance_sampling_ratio/mean": 1.0093098878860474, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9477690681815147, "clip_ratio/low_mean": 0.07486437633633614, "clip_ratio/low_min": 0.07486437633633614, "clip_ratio/high_mean": 0.04523809626698494, "clip_ratio/high_max": 0.04523809626698494, "clip_ratio/region_mean": 0.12010247260332108, "reward_total_mean": 0.8676860332489014, "reward_meter_mean": 0.9893183708190918, "reward_meter_std": 0.011413329280912876, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9586776494979858, "reward_repeat_soft_std": 0.03852525353431702, "reward_judge_quality_mean": 0.5887500047683716, "reward_judge_quality_std": 0.21596544981002808, "reward_total_composite_mean": 0.8676860332489014, "reward_total_composite_std": 0.06873012334108353} {"timestamp_utc": "2026-04-13T01:51:34Z", "mode": "train", "global_step": 1415, "epoch": 0.14213962832747362, "loss": -0.0047, "grad_norm": 12.262775421142578, "learning_rate": 5.715151515151516e-06, "num_tokens": 2538120.0, "completions/mean_length": 41.625, "completions/min_length": 38.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.625, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9633732438087463, "rewards/meter/std": 0.03487769141793251, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7755966186523438, "rewards/repeat_soft/std": 0.15112251043319702, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7904525995254517, "rewards/total_composite/std": 0.02551787719130516, "reward": 0.7904525995254517, "reward_std": 0.025517866015434265, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09970787167549133, "sampling/sampling_logp_difference/max": 1.221695899963379, "sampling/importance_sampling_ratio/min": 0.29472991824150085, "sampling/importance_sampling_ratio/mean": 1.0118688344955444, "sampling/importance_sampling_ratio/max": 1.9415382146835327, "entropy": 0.6051472201943398, "clip_ratio/low_mean": 0.04704698221758008, "clip_ratio/low_min": 0.04704698221758008, "clip_ratio/high_mean": 0.04346092604100704, "clip_ratio/high_max": 0.04346092604100704, "clip_ratio/region_mean": 0.09050790825858712, "reward_total_mean": 0.7904525995254517, "reward_meter_mean": 0.9633732438087463, "reward_meter_std": 0.03487769141793251, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7755966186523438, "reward_repeat_soft_std": 0.15112251043319702, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7904525995254517, "reward_total_composite_std": 0.02551787719130516} {"timestamp_utc": "2026-04-13T01:51:42Z", "mode": "train", "global_step": 1416, "epoch": 0.14224008036162733, "loss": 0.0112, "grad_norm": 14.434419631958008, "learning_rate": 5.712121212121212e-06, "num_tokens": 2539519.0, "completions/mean_length": 22.875, "completions/min_length": 21.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.875, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.9936314821243286, "rewards/meter/std": 0.0037631194572895765, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.929430365562439, "rewards/repeat_soft/std": 0.06355131417512894, "rewards/judge_quality/mean": 0.32624998688697815, "rewards/judge_quality/std": 0.11350739002227783, "rewards/total_composite/mean": 0.787952184677124, "rewards/total_composite/std": 0.03709360957145691, "reward": 0.787952184677124, "reward_std": 0.0370936244726181, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10948045551776886, "sampling/sampling_logp_difference/max": 2.2775824069976807, "sampling/importance_sampling_ratio/min": 0.1025317907333374, "sampling/importance_sampling_ratio/mean": 1.0423510074615479, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.81192646920681, "clip_ratio/low_mean": 0.05689102644100785, "clip_ratio/low_min": 0.05689102644100785, "clip_ratio/high_mean": 0.038334627635777, "clip_ratio/high_max": 0.038334627635777, "clip_ratio/region_mean": 0.09522565407678485, "reward_total_mean": 0.787952184677124, "reward_meter_mean": 0.9936314821243286, "reward_meter_std": 0.0037631194572895765, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.929430365562439, "reward_repeat_soft_std": 0.06355131417512894, "reward_judge_quality_mean": 0.32624998688697815, "reward_judge_quality_std": 0.11350739002227783, "reward_total_composite_mean": 0.787952184677124, "reward_total_composite_std": 0.03709360957145691} {"timestamp_utc": "2026-04-13T01:51:49Z", "mode": "train", "global_step": 1417, "epoch": 0.142340532395781, "loss": 0.0451, "grad_norm": 13.443899154663086, "learning_rate": 5.7090909090909096e-06, "num_tokens": 2540978.0, "completions/mean_length": 32.375, "completions/min_length": 28.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.375, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.7830171585083008, "rewards/meter/std": 0.2617934048175812, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9589030742645264, "rewards/repeat_soft/std": 0.08515104651451111, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.1865811049938202, "rewards/total_composite/mean": 0.7576230764389038, "rewards/total_composite/std": 0.12762218713760376, "reward": 0.7576230764389038, "reward_std": 0.12762217223644257, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10674096643924713, "sampling/sampling_logp_difference/max": 1.115617275238037, "sampling/importance_sampling_ratio/min": 0.32771292328834534, "sampling/importance_sampling_ratio/mean": 1.0040115118026733, "sampling/importance_sampling_ratio/max": 1.6143876314163208, "entropy": 0.6427342742681503, "clip_ratio/low_mean": 0.029953917488455772, "clip_ratio/low_min": 0.029953917488455772, "clip_ratio/high_mean": 0.09448713902384043, "clip_ratio/high_max": 0.09448713902384043, "clip_ratio/region_mean": 0.1244410565122962, "reward_total_mean": 0.7576230764389038, "reward_meter_mean": 0.7830171585083008, "reward_meter_std": 0.2617934048175812, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9589030742645264, "reward_repeat_soft_std": 0.08515104651451111, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.1865811049938202, "reward_total_composite_mean": 0.7576230764389038, "reward_total_composite_std": 0.12762218713760376} {"timestamp_utc": "2026-04-13T01:51:57Z", "mode": "train", "global_step": 1418, "epoch": 0.14244098442993472, "loss": 0.0212, "grad_norm": 10.386975288391113, "learning_rate": 5.706060606060606e-06, "num_tokens": 2542556.0, "completions/mean_length": 46.25, "completions/min_length": 40.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.25, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.9600078463554382, "rewards/meter/std": 0.07958407700061798, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.955445408821106, "rewards/repeat_soft/std": 0.031246360391378403, "rewards/judge_quality/mean": 0.44999998807907104, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.812548041343689, "rewards/total_composite/std": 0.03509565815329552, "reward": 0.812548041343689, "reward_std": 0.03509565442800522, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1236305683851242, "sampling/sampling_logp_difference/max": 2.8511695861816406, "sampling/importance_sampling_ratio/min": 0.05777670443058014, "sampling/importance_sampling_ratio/mean": 0.9991981387138367, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8163572549819946, "clip_ratio/low_mean": 0.01595744676887989, "clip_ratio/low_min": 0.01595744676887989, "clip_ratio/high_mean": 0.1124369150493294, "clip_ratio/high_max": 0.1124369150493294, "clip_ratio/region_mean": 0.1283943618182093, "reward_total_mean": 0.812548041343689, "reward_meter_mean": 0.9600078463554382, "reward_meter_std": 0.07958407700061798, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.955445408821106, "reward_repeat_soft_std": 0.031246360391378403, "reward_judge_quality_mean": 0.44999998807907104, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.812548041343689, "reward_total_composite_std": 0.03509565815329552} {"timestamp_utc": "2026-04-13T01:52:04Z", "mode": "train", "global_step": 1419, "epoch": 0.1425414364640884, "loss": 0.0155, "grad_norm": 12.338130950927734, "learning_rate": 5.703030303030303e-06, "num_tokens": 2544122.0, "completions/mean_length": 21.75, "completions/min_length": 20.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.75, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.9666792154312134, "rewards/meter/std": 0.036084845662117004, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.802371084690094, "rewards/repeat_soft/std": 0.16167213022708893, "rewards/judge_quality/mean": 0.4762499928474426, "rewards/judge_quality/std": 0.19167961180210114, "rewards/total_composite/mean": 0.8081177473068237, "rewards/total_composite/std": 0.06848397850990295, "reward": 0.8081177473068237, "reward_std": 0.06848399341106415, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15935446321964264, "sampling/sampling_logp_difference/max": 2.5440492630004883, "sampling/importance_sampling_ratio/min": 0.07854770123958588, "sampling/importance_sampling_ratio/mean": 1.0008351802825928, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4548356607556343, "clip_ratio/low_mean": 0.06427865754812956, "clip_ratio/low_min": 0.06427865754812956, "clip_ratio/high_mean": 0.04454051377251744, "clip_ratio/high_max": 0.04454051377251744, "clip_ratio/region_mean": 0.108819171320647, "reward_total_mean": 0.8081177473068237, "reward_meter_mean": 0.9666792154312134, "reward_meter_std": 0.036084845662117004, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.802371084690094, "reward_repeat_soft_std": 0.16167213022708893, "reward_judge_quality_mean": 0.4762499928474426, "reward_judge_quality_std": 0.19167961180210114, "reward_total_composite_mean": 0.8081177473068237, "reward_total_composite_std": 0.06848397850990295} {"timestamp_utc": "2026-04-13T01:52:12Z", "mode": "train", "global_step": 1420, "epoch": 0.14264188849824208, "loss": 0.0076, "grad_norm": 11.47413158416748, "learning_rate": 5.7e-06, "num_tokens": 2545704.0, "completions/mean_length": 43.75, "completions/min_length": 42.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.75, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.9815000295639038, "rewards/meter/std": 0.015268692746758461, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.836102306842804, "rewards/repeat_soft/std": 0.07281187176704407, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.8057852387428284, "rewards/total_composite/std": 0.014390409924089909, "reward": 0.8057852387428284, "reward_std": 0.014390409924089909, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09322912991046906, "sampling/sampling_logp_difference/max": 1.236918330192566, "sampling/importance_sampling_ratio/min": 0.2902773916721344, "sampling/importance_sampling_ratio/mean": 1.0032892227172852, "sampling/importance_sampling_ratio/max": 1.900910496711731, "entropy": 0.5646492093801498, "clip_ratio/low_mean": 0.03179112635552883, "clip_ratio/low_min": 0.03179112635552883, "clip_ratio/high_mean": 0.059547885321080685, "clip_ratio/high_max": 0.059547885321080685, "clip_ratio/region_mean": 0.09133901167660952, "reward_total_mean": 0.8057852387428284, "reward_meter_mean": 0.9815000295639038, "reward_meter_std": 0.015268692746758461, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.836102306842804, "reward_repeat_soft_std": 0.07281187176704407, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.8057852387428284, "reward_total_composite_std": 0.014390409924089909} {"timestamp_utc": "2026-04-13T01:52:19Z", "mode": "train", "global_step": 1421, "epoch": 0.1427423405323958, "loss": 0.0516, "grad_norm": 15.765979766845703, "learning_rate": 5.696969696969698e-06, "num_tokens": 2547050.0, "completions/mean_length": 26.25, "completions/min_length": 23.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.25, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.8893177509307861, "rewards/meter/std": 0.21628688275814056, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9393099546432495, "rewards/repeat_soft/std": 0.05938025936484337, "rewards/judge_quality/mean": 0.5475000143051147, "rewards/judge_quality/std": 0.3311344087123871, "rewards/total_composite/mean": 0.8083739876747131, "rewards/total_composite/std": 0.10878659784793854, "reward": 0.8083739876747131, "reward_std": 0.10878661274909973, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13830779492855072, "sampling/sampling_logp_difference/max": 1.2384634017944336, "sampling/importance_sampling_ratio/min": 0.28982922434806824, "sampling/importance_sampling_ratio/mean": 1.0127723217010498, "sampling/importance_sampling_ratio/max": 1.823634147644043, "entropy": 0.9909995645284653, "clip_ratio/low_mean": 0.029245014302432537, "clip_ratio/low_min": 0.029245014302432537, "clip_ratio/high_mean": 0.08452321402728558, "clip_ratio/high_max": 0.08452321402728558, "clip_ratio/region_mean": 0.11376822832971811, "reward_total_mean": 0.8083739876747131, "reward_meter_mean": 0.8893177509307861, "reward_meter_std": 0.21628688275814056, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9393099546432495, "reward_repeat_soft_std": 0.05938025936484337, "reward_judge_quality_mean": 0.5475000143051147, "reward_judge_quality_std": 0.3311344087123871, "reward_total_composite_mean": 0.8083739876747131, "reward_total_composite_std": 0.10878659784793854} {"timestamp_utc": "2026-04-13T01:52:26Z", "mode": "train", "global_step": 1422, "epoch": 0.14284279256654947, "loss": 0.024, "grad_norm": 8.226017951965332, "learning_rate": 5.693939393939394e-06, "num_tokens": 2549158.0, "completions/mean_length": 94.5, "completions/min_length": 78.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.5, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.961037278175354, "rewards/meter/std": 0.07393273711204529, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9227004647254944, "rewards/repeat_soft/std": 0.05491863563656807, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.7842368483543396, "rewards/total_composite/std": 0.042212676256895065, "reward": 0.7842368483543396, "reward_std": 0.04221266880631447, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12655109167099, "sampling/sampling_logp_difference/max": 1.2547664642333984, "sampling/importance_sampling_ratio/min": 0.2851424217224121, "sampling/importance_sampling_ratio/mean": 1.0223135948181152, "sampling/importance_sampling_ratio/max": 1.8440438508987427, "entropy": 1.2693314850330353, "clip_ratio/low_mean": 0.0392227191478014, "clip_ratio/low_min": 0.0392227191478014, "clip_ratio/high_mean": 0.06961366534233093, "clip_ratio/high_max": 0.06961366534233093, "clip_ratio/region_mean": 0.10883638449013233, "reward_total_mean": 0.7842368483543396, "reward_meter_mean": 0.961037278175354, "reward_meter_std": 0.07393273711204529, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9227004647254944, "reward_repeat_soft_std": 0.05491863563656807, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.7842368483543396, "reward_total_composite_std": 0.042212676256895065} {"timestamp_utc": "2026-04-13T01:52:33Z", "mode": "train", "global_step": 1423, "epoch": 0.14294324460070315, "loss": 0.0147, "grad_norm": 6.970751762390137, "learning_rate": 5.690909090909091e-06, "num_tokens": 2551694.0, "completions/mean_length": 126.0, "completions/min_length": 115.0, "completions/max_length": 137.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.0, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.8314439058303833, "rewards/meter/std": 0.24047192931175232, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8165693879127502, "rewards/repeat_soft/std": 0.0534350723028183, "rewards/judge_quality/mean": 0.21250000596046448, "rewards/judge_quality/std": 0.0517549142241478, "rewards/total_composite/mean": 0.6695566773414612, "rewards/total_composite/std": 0.10650607943534851, "reward": 0.6695566773414612, "reward_std": 0.10650607943534851, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11156121641397476, "sampling/sampling_logp_difference/max": 1.526178002357483, "sampling/importance_sampling_ratio/min": 0.2173648625612259, "sampling/importance_sampling_ratio/mean": 1.0178383588790894, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8699766248464584, "clip_ratio/low_mean": 0.028398494236171246, "clip_ratio/low_min": 0.028398494236171246, "clip_ratio/high_mean": 0.06029798090457916, "clip_ratio/high_max": 0.06029798090457916, "clip_ratio/region_mean": 0.08869647514075041, "reward_total_mean": 0.6695566773414612, "reward_meter_mean": 0.8314439058303833, "reward_meter_std": 0.24047192931175232, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8165693879127502, "reward_repeat_soft_std": 0.0534350723028183, "reward_judge_quality_mean": 0.21250000596046448, "reward_judge_quality_std": 0.0517549142241478, "reward_total_composite_mean": 0.6695566773414612, "reward_total_composite_std": 0.10650607943534851} {"timestamp_utc": "2026-04-13T01:52:41Z", "mode": "train", "global_step": 1424, "epoch": 0.14304369663485686, "loss": 0.0203, "grad_norm": 7.663969039916992, "learning_rate": 5.687878787878789e-06, "num_tokens": 2554018.0, "completions/mean_length": 100.5, "completions/min_length": 87.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.5, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.9525831937789917, "rewards/meter/std": 0.1128593236207962, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9005595445632935, "rewards/repeat_soft/std": 0.0479399748146534, "rewards/judge_quality/mean": 0.3512499928474426, "rewards/judge_quality/std": 0.16762521862983704, "rewards/total_composite/mean": 0.7740933895111084, "rewards/total_composite/std": 0.08158931136131287, "reward": 0.7740933895111084, "reward_std": 0.08158931881189346, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12350748479366302, "sampling/sampling_logp_difference/max": 1.500687599182129, "sampling/importance_sampling_ratio/min": 0.22297680377960205, "sampling/importance_sampling_ratio/mean": 1.0193243026733398, "sampling/importance_sampling_ratio/max": 1.809526801109314, "entropy": 1.0392766669392586, "clip_ratio/low_mean": 0.057389311492443085, "clip_ratio/low_min": 0.057389311492443085, "clip_ratio/high_mean": 0.0353688495233655, "clip_ratio/high_max": 0.0353688495233655, "clip_ratio/region_mean": 0.09275816101580858, "reward_total_mean": 0.7740933895111084, "reward_meter_mean": 0.9525831937789917, "reward_meter_std": 0.1128593236207962, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9005595445632935, "reward_repeat_soft_std": 0.0479399748146534, "reward_judge_quality_mean": 0.3512499928474426, "reward_judge_quality_std": 0.16762521862983704, "reward_total_composite_mean": 0.7740933895111084, "reward_total_composite_std": 0.08158931136131287} {"timestamp_utc": "2026-04-13T01:52:53Z", "mode": "train", "global_step": 1425, "epoch": 0.14314414866901054, "loss": -0.1369, "grad_norm": 4.212625026702881, "learning_rate": 5.684848484848485e-06, "num_tokens": 2556274.0, "completions/mean_length": 145.0, "completions/min_length": 78.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 92.5714340209961, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.47143641114234924, "rewards/meter/std": 0.3938138782978058, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.18898223340511322, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9342879056930542, "rewards/repeat_soft/std": 0.03885430097579956, "rewards/judge_quality/mean": 0.3187499940395355, "rewards/judge_quality/std": 0.14961257576942444, "rewards/total_composite/mean": 0.4528884291648865, "rewards/total_composite/std": 0.2547700107097626, "reward": 0.4528884291648865, "reward_std": 0.2547700107097626, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1332603096961975, "sampling/sampling_logp_difference/max": 1.6884803771972656, "sampling/importance_sampling_ratio/min": 0.1848001331090927, "sampling/importance_sampling_ratio/mean": 1.0246182680130005, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.059845395386219, "clip_ratio/low_mean": 0.024527186527848244, "clip_ratio/low_min": 0.024527186527848244, "clip_ratio/high_mean": 0.081673345528543, "clip_ratio/high_max": 0.081673345528543, "clip_ratio/region_mean": 0.10620053205639124, "reward_total_mean": 0.4528884291648865, "reward_meter_mean": 0.47143641114234924, "reward_meter_std": 0.3938138782978058, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.18898223340511322, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9342879056930542, "reward_repeat_soft_std": 0.03885430097579956, "reward_judge_quality_mean": 0.3187499940395355, "reward_judge_quality_std": 0.14961257576942444, "reward_total_composite_mean": 0.4528884291648865, "reward_total_composite_std": 0.2547700107097626} {"timestamp_utc": "2026-04-13T01:53:00Z", "mode": "train", "global_step": 1426, "epoch": 0.14324460070316425, "loss": 0.0466, "grad_norm": 13.343033790588379, "learning_rate": 5.681818181818183e-06, "num_tokens": 2557791.0, "completions/mean_length": 40.625, "completions/min_length": 38.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.625, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.955473780632019, "rewards/meter/std": 0.03899914398789406, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8996809720993042, "rewards/repeat_soft/std": 0.05421128869056702, "rewards/judge_quality/mean": 0.42250001430511475, "rewards/judge_quality/std": 0.15471172332763672, "rewards/total_composite/mean": 0.79668128490448, "rewards/total_composite/std": 0.05858469381928444, "reward": 0.79668128490448, "reward_std": 0.05858468636870384, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11397376656532288, "sampling/sampling_logp_difference/max": 1.5268630981445312, "sampling/importance_sampling_ratio/min": 0.21721599996089935, "sampling/importance_sampling_ratio/mean": 1.017112374305725, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7752817794680595, "clip_ratio/low_mean": 0.03723448282107711, "clip_ratio/low_min": 0.03723448282107711, "clip_ratio/high_mean": 0.030813954304903746, "clip_ratio/high_max": 0.030813954304903746, "clip_ratio/region_mean": 0.06804843712598085, "reward_total_mean": 0.79668128490448, "reward_meter_mean": 0.955473780632019, "reward_meter_std": 0.03899914398789406, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8996809720993042, "reward_repeat_soft_std": 0.05421128869056702, "reward_judge_quality_mean": 0.42250001430511475, "reward_judge_quality_std": 0.15471172332763672, "reward_total_composite_mean": 0.79668128490448, "reward_total_composite_std": 0.05858469381928444} {"timestamp_utc": "2026-04-13T01:53:07Z", "mode": "train", "global_step": 1427, "epoch": 0.14334505273731793, "loss": 0.0269, "grad_norm": 13.151548385620117, "learning_rate": 5.67878787878788e-06, "num_tokens": 2559441.0, "completions/mean_length": 46.25, "completions/min_length": 43.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.25, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.9738712310791016, "rewards/meter/std": 0.04348565638065338, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9520028829574585, "rewards/repeat_soft/std": 0.05743032693862915, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8116923570632935, "rewards/total_composite/std": 0.020018745213747025, "reward": 0.8116923570632935, "reward_std": 0.02001875452697277, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15186218917369843, "sampling/sampling_logp_difference/max": 1.427072525024414, "sampling/importance_sampling_ratio/min": 0.24001051485538483, "sampling/importance_sampling_ratio/mean": 1.0055372714996338, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2631877288222313, "clip_ratio/low_mean": 0.03243371192365885, "clip_ratio/low_min": 0.03243371192365885, "clip_ratio/high_mean": 0.14130384102463722, "clip_ratio/high_max": 0.14130384102463722, "clip_ratio/region_mean": 0.17373755294829607, "reward_total_mean": 0.8116923570632935, "reward_meter_mean": 0.9738712310791016, "reward_meter_std": 0.04348565638065338, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9520028829574585, "reward_repeat_soft_std": 0.05743032693862915, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8116923570632935, "reward_total_composite_std": 0.020018745213747025} {"timestamp_utc": "2026-04-13T01:53:14Z", "mode": "train", "global_step": 1428, "epoch": 0.1434455047714716, "loss": 0.0537, "grad_norm": 9.339617729187012, "learning_rate": 5.675757575757577e-06, "num_tokens": 2561552.0, "completions/mean_length": 84.875, "completions/min_length": 77.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.875, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.6577740907669067, "rewards/meter/std": 0.4543357789516449, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8631981611251831, "rewards/repeat_soft/std": 0.06688826531171799, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.6481931209564209, "rewards/total_composite/std": 0.22071678936481476, "reward": 0.6481931209564209, "reward_std": 0.22071677446365356, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11671317368745804, "sampling/sampling_logp_difference/max": 1.1993942260742188, "sampling/importance_sampling_ratio/min": 0.30137673020362854, "sampling/importance_sampling_ratio/mean": 1.0284929275512695, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0108622461557388, "clip_ratio/low_mean": 0.04117477685213089, "clip_ratio/low_min": 0.04117477685213089, "clip_ratio/high_mean": 0.07313114404678345, "clip_ratio/high_max": 0.07313114404678345, "clip_ratio/region_mean": 0.11430592089891434, "reward_total_mean": 0.6481931209564209, "reward_meter_mean": 0.6577740907669067, "reward_meter_std": 0.4543357789516449, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8631981611251831, "reward_repeat_soft_std": 0.06688826531171799, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.6481931209564209, "reward_total_composite_std": 0.22071678936481476} {"timestamp_utc": "2026-04-13T01:53:25Z", "mode": "train", "global_step": 1429, "epoch": 0.14354595680562532, "loss": -0.1736, "grad_norm": 2.163102626800537, "learning_rate": 5.672727272727273e-06, "num_tokens": 2563536.0, "completions/mean_length": 124.0, "completions/min_length": 63.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 68.5714340209961, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.8355300426483154, "rewards/meter/std": 0.34561896324157715, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9475846886634827, "rewards/repeat_soft/std": 0.03877967596054077, "rewards/judge_quality/mean": 0.2887499928474426, "rewards/judge_quality/std": 0.12799972295761108, "rewards/total_composite/mean": 0.6742470264434814, "rewards/total_composite/std": 0.2744044363498688, "reward": 0.6742470264434814, "reward_std": 0.2744044363498688, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1334671974182129, "sampling/sampling_logp_difference/max": 0.9917171001434326, "sampling/importance_sampling_ratio/min": 0.3709391951560974, "sampling/importance_sampling_ratio/mean": 1.0373542308807373, "sampling/importance_sampling_ratio/max": 1.796446681022644, "entropy": 1.2279027700424194, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.11141906306147575, "clip_ratio/high_max": 0.11141906306147575, "clip_ratio/region_mean": 0.11141906306147575, "reward_total_mean": 0.6742470264434814, "reward_meter_mean": 0.8355300426483154, "reward_meter_std": 0.34561896324157715, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9475846886634827, "reward_repeat_soft_std": 0.03877967596054077, "reward_judge_quality_mean": 0.2887499928474426, "reward_judge_quality_std": 0.12799972295761108, "reward_total_composite_mean": 0.6742470264434814, "reward_total_composite_std": 0.2744044363498688} {"timestamp_utc": "2026-04-13T01:53:38Z", "mode": "train", "global_step": 1430, "epoch": 0.143646408839779, "loss": -0.1304, "grad_norm": 1.6232706308364868, "learning_rate": 5.6696969696969705e-06, "num_tokens": 2565294.0, "completions/mean_length": 101.75, "completions/min_length": 39.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 43.142860412597656, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.9507706165313721, "rewards/meter/std": 0.09293875843286514, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9484732151031494, "rewards/repeat_soft/std": 0.05257227644324303, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.1403057724237442, "rewards/total_composite/mean": 0.6995062828063965, "rewards/total_composite/std": 0.2835659980773926, "reward": 0.6995062828063965, "reward_std": 0.2835659980773926, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12631267309188843, "sampling/sampling_logp_difference/max": 1.3627243041992188, "sampling/importance_sampling_ratio/min": 0.2559624910354614, "sampling/importance_sampling_ratio/mean": 1.0080418586730957, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9766185730695724, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.12111808825284243, "clip_ratio/high_max": 0.12111808825284243, "clip_ratio/region_mean": 0.12111808825284243, "reward_total_mean": 0.6995062828063965, "reward_meter_mean": 0.9507706165313721, "reward_meter_std": 0.09293875843286514, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9484732151031494, "reward_repeat_soft_std": 0.05257227644324303, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.1403057724237442, "reward_total_composite_mean": 0.6995062828063965, "reward_total_composite_std": 0.2835659980773926} {"timestamp_utc": "2026-04-13T01:53:53Z", "mode": "train", "global_step": 1431, "epoch": 0.1437468608739327, "loss": -0.089, "grad_norm": 2.071838855743408, "learning_rate": 5.666666666666667e-06, "num_tokens": 2566735.0, "completions/mean_length": 160.125, "completions/min_length": 39.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 42.833335876464844, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.6227982640266418, "rewards/meter/std": 0.5005600452423096, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9415609240531921, "rewards/repeat_soft/std": 0.042996373027563095, "rewards/judge_quality/mean": 0.2150000035762787, "rewards/judge_quality/std": 0.13180071115493774, "rewards/total_composite/mean": 0.5468136072158813, "rewards/total_composite/std": 0.3250689208507538, "reward": 0.5468136072158813, "reward_std": 0.3250689208507538, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13953259587287903, "sampling/sampling_logp_difference/max": 1.3585014343261719, "sampling/importance_sampling_ratio/min": 0.2570456862449646, "sampling/importance_sampling_ratio/mean": 1.0595859289169312, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0005056709051132, "clip_ratio/low_mean": 0.010204081423580647, "clip_ratio/low_min": 0.010204081423580647, "clip_ratio/high_mean": 0.05374050512909889, "clip_ratio/high_max": 0.05374050512909889, "clip_ratio/region_mean": 0.06394458655267954, "reward_total_mean": 0.5468136072158813, "reward_meter_mean": 0.6227982640266418, "reward_meter_std": 0.5005600452423096, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9415609240531921, "reward_repeat_soft_std": 0.042996373027563095, "reward_judge_quality_mean": 0.2150000035762787, "reward_judge_quality_std": 0.13180071115493774, "reward_total_composite_mean": 0.5468136072158813, "reward_total_composite_std": 0.3250689208507538} {"timestamp_utc": "2026-04-13T01:54:05Z", "mode": "train", "global_step": 1432, "epoch": 0.1438473129080864, "loss": -0.1679, "grad_norm": 2.417551040649414, "learning_rate": 5.663636363636364e-06, "num_tokens": 2569133.0, "completions/mean_length": 165.75, "completions/min_length": 109.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 116.28572082519531, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.640867292881012, "rewards/meter/std": 0.3843971788883209, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.786146879196167, "rewards/repeat_soft/std": 0.18548940122127533, "rewards/judge_quality/mean": 0.19624999165534973, "rewards/judge_quality/std": 0.11070391535758972, "rewards/total_composite/mean": 0.4871726930141449, "rewards/total_composite/std": 0.23759232461452484, "reward": 0.4871726930141449, "reward_std": 0.23759230971336365, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09746279567480087, "sampling/sampling_logp_difference/max": 2.1929824352264404, "sampling/importance_sampling_ratio/min": 0.11158346384763718, "sampling/importance_sampling_ratio/mean": 1.0131382942199707, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5286736860871315, "clip_ratio/low_mean": 0.033638945780694485, "clip_ratio/low_min": 0.033638945780694485, "clip_ratio/high_mean": 0.03125028661452234, "clip_ratio/high_max": 0.03125028661452234, "clip_ratio/region_mean": 0.06488923239521682, "reward_total_mean": 0.4871726930141449, "reward_meter_mean": 0.640867292881012, "reward_meter_std": 0.3843971788883209, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.786146879196167, "reward_repeat_soft_std": 0.18548940122127533, "reward_judge_quality_mean": 0.19624999165534973, "reward_judge_quality_std": 0.11070391535758972, "reward_total_composite_mean": 0.4871726930141449, "reward_total_composite_std": 0.23759232461452484} {"timestamp_utc": "2026-04-13T01:54:11Z", "mode": "train", "global_step": 1433, "epoch": 0.14394776494224007, "loss": 0.0036, "grad_norm": 10.0048246383667, "learning_rate": 5.6606060606060606e-06, "num_tokens": 2570782.0, "completions/mean_length": 45.125, "completions/min_length": 42.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.125, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.9552374482154846, "rewards/meter/std": 0.032310791313648224, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7891312837600708, "rewards/repeat_soft/std": 0.07586158812046051, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7720199823379517, "rewards/total_composite/std": 0.022065138444304466, "reward": 0.7720199823379517, "reward_std": 0.022065138444304466, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11640465259552002, "sampling/sampling_logp_difference/max": 1.7247743606567383, "sampling/importance_sampling_ratio/min": 0.17821326851844788, "sampling/importance_sampling_ratio/mean": 1.0081477165222168, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8101698234677315, "clip_ratio/low_mean": 0.03414972685277462, "clip_ratio/low_min": 0.03414972685277462, "clip_ratio/high_mean": 0.09066418930888176, "clip_ratio/high_max": 0.09066418930888176, "clip_ratio/region_mean": 0.12481391616165638, "reward_total_mean": 0.7720199823379517, "reward_meter_mean": 0.9552374482154846, "reward_meter_std": 0.032310791313648224, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7891312837600708, "reward_repeat_soft_std": 0.07586158812046051, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7720199823379517, "reward_total_composite_std": 0.022065138444304466} {"timestamp_utc": "2026-04-13T01:54:19Z", "mode": "train", "global_step": 1434, "epoch": 0.14404821697639378, "loss": -0.0392, "grad_norm": 10.55432415008545, "learning_rate": 5.657575757575759e-06, "num_tokens": 2572158.0, "completions/mean_length": 48.0, "completions/min_length": 39.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.0, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9753754138946533, "rewards/meter/std": 0.026589160785079002, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9531626105308533, "rewards/repeat_soft/std": 0.05708824470639229, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8102352023124695, "rewards/total_composite/std": 0.013530041091144085, "reward": 0.8102352023124695, "reward_std": 0.013530049473047256, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12738707661628723, "sampling/sampling_logp_difference/max": 1.6169452667236328, "sampling/importance_sampling_ratio/min": 0.19850414991378784, "sampling/importance_sampling_ratio/mean": 1.0360651016235352, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1591717898845673, "clip_ratio/low_mean": 0.018138112500309944, "clip_ratio/low_min": 0.018138112500309944, "clip_ratio/high_mean": 0.0756379752419889, "clip_ratio/high_max": 0.0756379752419889, "clip_ratio/region_mean": 0.09377608774229884, "reward_total_mean": 0.8102352023124695, "reward_meter_mean": 0.9753754138946533, "reward_meter_std": 0.026589160785079002, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9531626105308533, "reward_repeat_soft_std": 0.05708824470639229, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8102352023124695, "reward_total_composite_std": 0.013530041091144085} {"timestamp_utc": "2026-04-13T01:54:31Z", "mode": "train", "global_step": 1435, "epoch": 0.14414866901054746, "loss": -0.1024, "grad_norm": 3.082141160964966, "learning_rate": 5.654545454545455e-06, "num_tokens": 2573757.0, "completions/mean_length": 104.875, "completions/min_length": 43.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 46.71428680419922, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.8602102398872375, "rewards/meter/std": 0.3472323417663574, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9792897701263428, "rewards/repeat_soft/std": 0.012552016414701939, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.13617216050624847, "rewards/total_composite/mean": 0.6616296768188477, "rewards/total_composite/std": 0.30962860584259033, "reward": 0.6616296768188477, "reward_std": 0.30962857604026794, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11658137291669846, "sampling/sampling_logp_difference/max": 1.0877761840820312, "sampling/importance_sampling_ratio/min": 0.33696502447128296, "sampling/importance_sampling_ratio/mean": 1.016696572303772, "sampling/importance_sampling_ratio/max": 1.6495133638381958, "entropy": 0.7593571990728378, "clip_ratio/low_mean": 0.019607843831181526, "clip_ratio/low_min": 0.019607843831181526, "clip_ratio/high_mean": 0.07481290958821774, "clip_ratio/high_max": 0.07481290958821774, "clip_ratio/region_mean": 0.09442075341939926, "reward_total_mean": 0.6616296768188477, "reward_meter_mean": 0.8602102398872375, "reward_meter_std": 0.3472323417663574, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9792897701263428, "reward_repeat_soft_std": 0.012552016414701939, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.13617216050624847, "reward_total_composite_mean": 0.6616296768188477, "reward_total_composite_std": 0.30962860584259033} {"timestamp_utc": "2026-04-13T01:54:37Z", "mode": "train", "global_step": 1436, "epoch": 0.14424912104470117, "loss": 0.0594, "grad_norm": 26.381044387817383, "learning_rate": 5.651515151515152e-06, "num_tokens": 2575177.0, "completions/mean_length": 21.5, "completions/min_length": 16.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.5, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9782759547233582, "rewards/meter/std": 0.01913222298026085, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9354690909385681, "rewards/repeat_soft/std": 0.04015655070543289, "rewards/judge_quality/mean": 0.8025000095367432, "rewards/judge_quality/std": 0.21756774187088013, "rewards/total_composite/mean": 0.9245210886001587, "rewards/total_composite/std": 0.06235329061746597, "reward": 0.9245210886001587, "reward_std": 0.06235329061746597, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1113007441163063, "sampling/sampling_logp_difference/max": 1.2316102981567383, "sampling/importance_sampling_ratio/min": 0.29182225465774536, "sampling/importance_sampling_ratio/mean": 1.0135345458984375, "sampling/importance_sampling_ratio/max": 1.9345687627792358, "entropy": 0.8145139515399933, "clip_ratio/low_mean": 0.015625, "clip_ratio/low_min": 0.015625, "clip_ratio/high_mean": 0.08020683098584414, "clip_ratio/high_max": 0.08020683098584414, "clip_ratio/region_mean": 0.09583183098584414, "reward_total_mean": 0.9245210886001587, "reward_meter_mean": 0.9782759547233582, "reward_meter_std": 0.01913222298026085, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9354690909385681, "reward_repeat_soft_std": 0.04015655070543289, "reward_judge_quality_mean": 0.8025000095367432, "reward_judge_quality_std": 0.21756774187088013, "reward_total_composite_mean": 0.9245210886001587, "reward_total_composite_std": 0.06235329061746597} {"timestamp_utc": "2026-04-13T01:54:48Z", "mode": "train", "global_step": 1437, "epoch": 0.14434957307885485, "loss": -0.05, "grad_norm": 5.53255558013916, "learning_rate": 5.648484848484849e-06, "num_tokens": 2576742.0, "completions/mean_length": 101.625, "completions/min_length": 33.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 43.000003814697266, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.6529897451400757, "rewards/meter/std": 0.4121864438056946, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9622889161109924, "rewards/repeat_soft/std": 0.059122972190380096, "rewards/judge_quality/mean": 0.45249998569488525, "rewards/judge_quality/std": 0.2665520906448364, "rewards/total_composite/mean": 0.6015030145645142, "rewards/total_composite/std": 0.3819020092487335, "reward": 0.6015030145645142, "reward_std": 0.38190197944641113, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17184680700302124, "sampling/sampling_logp_difference/max": 1.233717918395996, "sampling/importance_sampling_ratio/min": 0.291207879781723, "sampling/importance_sampling_ratio/mean": 1.0056853294372559, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4514721110463142, "clip_ratio/low_mean": 0.017500000074505806, "clip_ratio/low_min": 0.017500000074505806, "clip_ratio/high_mean": 0.13004685379564762, "clip_ratio/high_max": 0.13004685379564762, "clip_ratio/region_mean": 0.14754685387015343, "reward_total_mean": 0.6015030145645142, "reward_meter_mean": 0.6529897451400757, "reward_meter_std": 0.4121864438056946, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9622889161109924, "reward_repeat_soft_std": 0.059122972190380096, "reward_judge_quality_mean": 0.45249998569488525, "reward_judge_quality_std": 0.2665520906448364, "reward_total_composite_mean": 0.6015030145645142, "reward_total_composite_std": 0.3819020092487335} {"timestamp_utc": "2026-04-13T01:54:56Z", "mode": "train", "global_step": 1438, "epoch": 0.14445002511300853, "loss": 0.0072, "grad_norm": 6.855283737182617, "learning_rate": 5.645454545454546e-06, "num_tokens": 2579555.0, "completions/mean_length": 135.625, "completions/min_length": 124.0, "completions/max_length": 152.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 135.625, "completions/min_terminated_length": 124.0, "completions/max_terminated_length": 152.0, "rewards/meter/mean": 0.7483280301094055, "rewards/meter/std": 0.312903493642807, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8217090368270874, "rewards/repeat_soft/std": 0.10505805164575577, "rewards/judge_quality/mean": 0.24249999225139618, "rewards/judge_quality/std": 0.11792854964733124, "rewards/total_composite/mean": 0.6416684985160828, "rewards/total_composite/std": 0.12375365942716599, "reward": 0.6416684985160828, "reward_std": 0.12375364452600479, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11121097952127457, "sampling/sampling_logp_difference/max": 1.8916023969650269, "sampling/importance_sampling_ratio/min": 0.15082992613315582, "sampling/importance_sampling_ratio/mean": 1.017565369606018, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8632097840309143, "clip_ratio/low_mean": 0.03592852782458067, "clip_ratio/low_min": 0.03592852782458067, "clip_ratio/high_mean": 0.07162978034466505, "clip_ratio/high_max": 0.07162978034466505, "clip_ratio/region_mean": 0.10755830816924572, "reward_total_mean": 0.6416684985160828, "reward_meter_mean": 0.7483280301094055, "reward_meter_std": 0.312903493642807, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8217090368270874, "reward_repeat_soft_std": 0.10505805164575577, "reward_judge_quality_mean": 0.24249999225139618, "reward_judge_quality_std": 0.11792854964733124, "reward_total_composite_mean": 0.6416684985160828, "reward_total_composite_std": 0.12375365942716599} {"timestamp_utc": "2026-04-13T01:55:03Z", "mode": "train", "global_step": 1439, "epoch": 0.14455047714716224, "loss": 0.0509, "grad_norm": 16.551481246948242, "learning_rate": 5.642424242424242e-06, "num_tokens": 2581412.0, "completions/mean_length": 70.125, "completions/min_length": 61.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.125, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.8735154867172241, "rewards/meter/std": 0.21798235177993774, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9797548055648804, "rewards/repeat_soft/std": 0.017108164727687836, "rewards/judge_quality/mean": 0.5575000047683716, "rewards/judge_quality/std": 0.19955310225486755, "rewards/total_composite/mean": 0.8083074688911438, "rewards/total_composite/std": 0.06645514816045761, "reward": 0.8083074688911438, "reward_std": 0.06645515561103821, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1724875569343567, "sampling/sampling_logp_difference/max": 1.8795347213745117, "sampling/importance_sampling_ratio/min": 0.15266112983226776, "sampling/importance_sampling_ratio/mean": 1.004686713218689, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8159292489290237, "clip_ratio/low_mean": 0.036472566425800323, "clip_ratio/low_min": 0.036472566425800323, "clip_ratio/high_mean": 0.11828634142875671, "clip_ratio/high_max": 0.11828634142875671, "clip_ratio/region_mean": 0.15475890785455704, "reward_total_mean": 0.8083074688911438, "reward_meter_mean": 0.8735154867172241, "reward_meter_std": 0.21798235177993774, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9797548055648804, "reward_repeat_soft_std": 0.017108164727687836, "reward_judge_quality_mean": 0.5575000047683716, "reward_judge_quality_std": 0.19955310225486755, "reward_total_composite_mean": 0.8083074688911438, "reward_total_composite_std": 0.06645514816045761} {"timestamp_utc": "2026-04-13T01:55:10Z", "mode": "train", "global_step": 1440, "epoch": 0.14465092918131592, "loss": -0.0294, "grad_norm": 5.796782970428467, "learning_rate": 5.6393939393939405e-06, "num_tokens": 2584015.0, "completions/mean_length": 120.375, "completions/min_length": 104.0, "completions/max_length": 137.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.375, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.9929150342941284, "rewards/meter/std": 0.00443782564252615, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7889989614486694, "rewards/repeat_soft/std": 0.09709357470273972, "rewards/judge_quality/mean": 0.2549999952316284, "rewards/judge_quality/std": 0.111867256462574, "rewards/total_composite/mean": 0.7484616637229919, "rewards/total_composite/std": 0.03810817003250122, "reward": 0.7484616637229919, "reward_std": 0.03810817748308182, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08997946977615356, "sampling/sampling_logp_difference/max": 1.6627349853515625, "sampling/importance_sampling_ratio/min": 0.18961966037750244, "sampling/importance_sampling_ratio/mean": 1.0077834129333496, "sampling/importance_sampling_ratio/max": 1.8250603675842285, "entropy": 0.7258725129067898, "clip_ratio/low_mean": 0.0325985848903656, "clip_ratio/low_min": 0.0325985848903656, "clip_ratio/high_mean": 0.034406693652272224, "clip_ratio/high_max": 0.034406693652272224, "clip_ratio/region_mean": 0.06700527854263783, "reward_total_mean": 0.7484616637229919, "reward_meter_mean": 0.9929150342941284, "reward_meter_std": 0.00443782564252615, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7889989614486694, "reward_repeat_soft_std": 0.09709357470273972, "reward_judge_quality_mean": 0.2549999952316284, "reward_judge_quality_std": 0.111867256462574, "reward_total_composite_mean": 0.7484616637229919, "reward_total_composite_std": 0.03810817003250122} {"timestamp_utc": "2026-04-13T01:55:17Z", "mode": "train", "global_step": 1441, "epoch": 0.14475138121546963, "loss": 0.0501, "grad_norm": 43.66619110107422, "learning_rate": 5.636363636363636e-06, "num_tokens": 2585607.0, "completions/mean_length": 41.0, "completions/min_length": 37.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.9599860906600952, "rewards/meter/std": 0.07541752606630325, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9713066816329956, "rewards/repeat_soft/std": 0.03700091689825058, "rewards/judge_quality/mean": 0.89000004529953, "rewards/judge_quality/std": 0.06164414808154106, "rewards/total_composite/mean": 0.9461244344711304, "rewards/total_composite/std": 0.05374912917613983, "reward": 0.9461244344711304, "reward_std": 0.05374912917613983, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11079558730125427, "sampling/sampling_logp_difference/max": 2.016066551208496, "sampling/importance_sampling_ratio/min": 0.13317829370498657, "sampling/importance_sampling_ratio/mean": 1.0133001804351807, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45956408604979515, "clip_ratio/low_mean": 0.02434593066573143, "clip_ratio/low_min": 0.02434593066573143, "clip_ratio/high_mean": 0.06434340309351683, "clip_ratio/high_max": 0.06434340309351683, "clip_ratio/region_mean": 0.08868933375924826, "reward_total_mean": 0.9461244344711304, "reward_meter_mean": 0.9599860906600952, "reward_meter_std": 0.07541752606630325, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9713066816329956, "reward_repeat_soft_std": 0.03700091689825058, "reward_judge_quality_mean": 0.89000004529953, "reward_judge_quality_std": 0.06164414808154106, "reward_total_composite_mean": 0.9461244344711304, "reward_total_composite_std": 0.05374912917613983} {"timestamp_utc": "2026-04-13T01:55:23Z", "mode": "train", "global_step": 1442, "epoch": 0.1448518332496233, "loss": 0.063, "grad_norm": 12.29006290435791, "learning_rate": 5.633333333333334e-06, "num_tokens": 2587069.0, "completions/mean_length": 43.75, "completions/min_length": 37.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.75, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9357615113258362, "rewards/meter/std": 0.059999242424964905, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9006228446960449, "rewards/repeat_soft/std": 0.08473695814609528, "rewards/judge_quality/mean": 0.4024999737739563, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.7819049954414368, "rewards/total_composite/std": 0.038997553288936615, "reward": 0.7819049954414368, "reward_std": 0.038997553288936615, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11360016465187073, "sampling/sampling_logp_difference/max": 0.9780607223510742, "sampling/importance_sampling_ratio/min": 0.37603962421417236, "sampling/importance_sampling_ratio/mean": 1.024580717086792, "sampling/importance_sampling_ratio/max": 1.9412214756011963, "entropy": 0.8131283521652222, "clip_ratio/low_mean": 0.0363151878118515, "clip_ratio/low_min": 0.0363151878118515, "clip_ratio/high_mean": 0.05872839782387018, "clip_ratio/high_max": 0.05872839782387018, "clip_ratio/region_mean": 0.09504358563572168, "reward_total_mean": 0.7819049954414368, "reward_meter_mean": 0.9357615113258362, "reward_meter_std": 0.059999242424964905, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9006228446960449, "reward_repeat_soft_std": 0.08473695814609528, "reward_judge_quality_mean": 0.4024999737739563, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.7819049954414368, "reward_total_composite_std": 0.038997553288936615} {"timestamp_utc": "2026-04-13T01:55:35Z", "mode": "train", "global_step": 1443, "epoch": 0.144952285283777, "loss": -0.177, "grad_norm": 1.7545239925384521, "learning_rate": 5.630303030303031e-06, "num_tokens": 2589068.0, "completions/mean_length": 130.875, "completions/min_length": 71.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 76.42857360839844, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.9899492263793945, "rewards/meter/std": 0.009538044221699238, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8823148012161255, "rewards/repeat_soft/std": 0.08612601459026337, "rewards/judge_quality/mean": 0.30124998092651367, "rewards/judge_quality/std": 0.14913439750671387, "rewards/total_composite/mean": 0.6852003335952759, "rewards/total_composite/std": 0.27885550260543823, "reward": 0.6852003335952759, "reward_std": 0.27885550260543823, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13998040556907654, "sampling/sampling_logp_difference/max": 2.1589272022247314, "sampling/importance_sampling_ratio/min": 0.11544890701770782, "sampling/importance_sampling_ratio/mean": 1.0126733779907227, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8198574557900429, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09049053117632866, "clip_ratio/high_max": 0.09049053117632866, "clip_ratio/region_mean": 0.09049053117632866, "reward_total_mean": 0.6852003335952759, "reward_meter_mean": 0.9899492263793945, "reward_meter_std": 0.009538044221699238, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8823148012161255, "reward_repeat_soft_std": 0.08612601459026337, "reward_judge_quality_mean": 0.30124998092651367, "reward_judge_quality_std": 0.14913439750671387, "reward_total_composite_mean": 0.6852003335952759, "reward_total_composite_std": 0.27885550260543823} {"timestamp_utc": "2026-04-13T01:55:43Z", "mode": "train", "global_step": 1444, "epoch": 0.1450527373179307, "loss": 0.0495, "grad_norm": 6.3622331619262695, "learning_rate": 5.627272727272728e-06, "num_tokens": 2591778.0, "completions/mean_length": 139.75, "completions/min_length": 128.0, "completions/max_length": 157.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 139.75, "completions/min_terminated_length": 128.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.8784022331237793, "rewards/meter/std": 0.12529930472373962, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8363226652145386, "rewards/repeat_soft/std": 0.07128876447677612, "rewards/judge_quality/mean": 0.2799999713897705, "rewards/judge_quality/std": 0.09304375946521759, "rewards/total_composite/mean": 0.7129132747650146, "rewards/total_composite/std": 0.06468817591667175, "reward": 0.7129132747650146, "reward_std": 0.06468819081783295, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0947369709610939, "sampling/sampling_logp_difference/max": 1.1861532926559448, "sampling/importance_sampling_ratio/min": 0.3053937554359436, "sampling/importance_sampling_ratio/mean": 1.0268645286560059, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7020629048347473, "clip_ratio/low_mean": 0.03636020328849554, "clip_ratio/low_min": 0.03636020328849554, "clip_ratio/high_mean": 0.04974737670272589, "clip_ratio/high_max": 0.04974737670272589, "clip_ratio/region_mean": 0.08610757999122143, "reward_total_mean": 0.7129132747650146, "reward_meter_mean": 0.8784022331237793, "reward_meter_std": 0.12529930472373962, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8363226652145386, "reward_repeat_soft_std": 0.07128876447677612, "reward_judge_quality_mean": 0.2799999713897705, "reward_judge_quality_std": 0.09304375946521759, "reward_total_composite_mean": 0.7129132747650146, "reward_total_composite_std": 0.06468817591667175} {"timestamp_utc": "2026-04-13T01:55:49Z", "mode": "train", "global_step": 1445, "epoch": 0.14515318935208438, "loss": 0.1073, "grad_norm": 17.18270492553711, "learning_rate": 5.624242424242424e-06, "num_tokens": 2593244.0, "completions/mean_length": 24.25, "completions/min_length": 18.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.25, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9810115098953247, "rewards/meter/std": 0.02146211452782154, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9448326230049133, "rewards/repeat_soft/std": 0.032767005264759064, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.8505634069442749, "rewards/total_composite/std": 0.06907282769680023, "reward": 0.8505634069442749, "reward_std": 0.06907281279563904, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1414811611175537, "sampling/sampling_logp_difference/max": 1.3768196105957031, "sampling/importance_sampling_ratio/min": 0.2523799538612366, "sampling/importance_sampling_ratio/mean": 1.0314865112304688, "sampling/importance_sampling_ratio/max": 1.7396936416625977, "entropy": 1.0603653825819492, "clip_ratio/low_mean": 0.13513986254110932, "clip_ratio/low_min": 0.13513986254110932, "clip_ratio/high_mean": 0.031746032647788525, "clip_ratio/high_max": 0.031746032647788525, "clip_ratio/region_mean": 0.16688589518889785, "reward_total_mean": 0.8505634069442749, "reward_meter_mean": 0.9810115098953247, "reward_meter_std": 0.02146211452782154, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9448326230049133, "reward_repeat_soft_std": 0.032767005264759064, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.8505634069442749, "reward_total_composite_std": 0.06907282769680023} {"timestamp_utc": "2026-04-13T01:56:00Z", "mode": "train", "global_step": 1446, "epoch": 0.14525364138623806, "loss": -0.1011, "grad_norm": 1.888236165046692, "learning_rate": 5.6212121212121215e-06, "num_tokens": 2594778.0, "completions/mean_length": 93.75, "completions/min_length": 29.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 34.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.7544448375701904, "rewards/meter/std": 0.3440179228782654, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.99051833152771, "rewards/repeat_soft/std": 0.015622739680111408, "rewards/judge_quality/mean": 0.4375, "rewards/judge_quality/std": 0.24662001430988312, "rewards/total_composite/mean": 0.6867468357086182, "rewards/total_composite/std": 0.29787012934684753, "reward": 0.6867468357086182, "reward_std": 0.29787009954452515, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11525560170412064, "sampling/sampling_logp_difference/max": 1.9637224674224854, "sampling/importance_sampling_ratio/min": 0.1403350532054901, "sampling/importance_sampling_ratio/mean": 0.996475338935852, "sampling/importance_sampling_ratio/max": 1.7051640748977661, "entropy": 0.47532327473163605, "clip_ratio/low_mean": 0.0069444444961845875, "clip_ratio/low_min": 0.0069444444961845875, "clip_ratio/high_mean": 0.07942771399393678, "clip_ratio/high_max": 0.07942771399393678, "clip_ratio/region_mean": 0.08637215849012136, "reward_total_mean": 0.6867468357086182, "reward_meter_mean": 0.7544448375701904, "reward_meter_std": 0.3440179228782654, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.99051833152771, "reward_repeat_soft_std": 0.015622739680111408, "reward_judge_quality_mean": 0.4375, "reward_judge_quality_std": 0.24662001430988312, "reward_total_composite_mean": 0.6867468357086182, "reward_total_composite_std": 0.29787012934684753} {"timestamp_utc": "2026-04-13T01:56:07Z", "mode": "train", "global_step": 1447, "epoch": 0.14535409342039177, "loss": 0.0052, "grad_norm": 11.450676918029785, "learning_rate": 5.618181818181818e-06, "num_tokens": 2596419.0, "completions/mean_length": 45.125, "completions/min_length": 41.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.125, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9927062392234802, "rewards/meter/std": 0.002369468566030264, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9202144145965576, "rewards/repeat_soft/std": 0.06686237454414368, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.10392305999994278, "rewards/total_composite/mean": 0.8282392621040344, "rewards/total_composite/std": 0.03417103365063667, "reward": 0.8282392621040344, "reward_std": 0.03417103737592697, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10443112999200821, "sampling/sampling_logp_difference/max": 1.331291913986206, "sampling/importance_sampling_ratio/min": 0.26413580775260925, "sampling/importance_sampling_ratio/mean": 1.017938256263733, "sampling/importance_sampling_ratio/max": 1.962907075881958, "entropy": 0.6326289959251881, "clip_ratio/low_mean": 0.08109235344454646, "clip_ratio/low_min": 0.08109235344454646, "clip_ratio/high_mean": 0.005434782709926367, "clip_ratio/high_max": 0.005434782709926367, "clip_ratio/region_mean": 0.08652713615447283, "reward_total_mean": 0.8282392621040344, "reward_meter_mean": 0.9927062392234802, "reward_meter_std": 0.002369468566030264, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9202144145965576, "reward_repeat_soft_std": 0.06686237454414368, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.10392305999994278, "reward_total_composite_mean": 0.8282392621040344, "reward_total_composite_std": 0.03417103365063667} {"timestamp_utc": "2026-04-13T01:56:14Z", "mode": "train", "global_step": 1448, "epoch": 0.14545454545454545, "loss": 0.0221, "grad_norm": 11.14900016784668, "learning_rate": 5.615151515151516e-06, "num_tokens": 2597883.0, "completions/mean_length": 36.0, "completions/min_length": 32.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9421075582504272, "rewards/meter/std": 0.05233609303832054, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9200968742370605, "rewards/repeat_soft/std": 0.10944931209087372, "rewards/judge_quality/mean": 0.5074999928474426, "rewards/judge_quality/std": 0.18077215552330017, "rewards/total_composite/mean": 0.8182080984115601, "rewards/total_composite/std": 0.06678342074155807, "reward": 0.8182080984115601, "reward_std": 0.06678341329097748, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10478964447975159, "sampling/sampling_logp_difference/max": 1.5408830642700195, "sampling/importance_sampling_ratio/min": 0.21419188380241394, "sampling/importance_sampling_ratio/mean": 1.0026323795318604, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.508757196366787, "clip_ratio/low_mean": 0.054673973005265, "clip_ratio/low_min": 0.054673973005265, "clip_ratio/high_mean": 0.03655462199822068, "clip_ratio/high_max": 0.03655462199822068, "clip_ratio/region_mean": 0.09122859500348568, "reward_total_mean": 0.8182080984115601, "reward_meter_mean": 0.9421075582504272, "reward_meter_std": 0.05233609303832054, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9200968742370605, "reward_repeat_soft_std": 0.10944931209087372, "reward_judge_quality_mean": 0.5074999928474426, "reward_judge_quality_std": 0.18077215552330017, "reward_total_composite_mean": 0.8182080984115601, "reward_total_composite_std": 0.06678342074155807} {"timestamp_utc": "2026-04-13T01:56:21Z", "mode": "train", "global_step": 1449, "epoch": 0.14555499748869916, "loss": -0.0098, "grad_norm": 12.979056358337402, "learning_rate": 5.612121212121212e-06, "num_tokens": 2599859.0, "completions/mean_length": 73.0, "completions/min_length": 55.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.6009846925735474, "rewards/meter/std": 0.31950291991233826, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8851910829544067, "rewards/repeat_soft/std": 0.056723177433013916, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.6184622049331665, "rewards/total_composite/std": 0.1360778957605362, "reward": 0.6184622049331665, "reward_std": 0.136077880859375, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12227115780115128, "sampling/sampling_logp_difference/max": 1.6753177642822266, "sampling/importance_sampling_ratio/min": 0.18724867701530457, "sampling/importance_sampling_ratio/mean": 1.007721185684204, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6777164563536644, "clip_ratio/low_mean": 0.05483816470950842, "clip_ratio/low_min": 0.05483816470950842, "clip_ratio/high_mean": 0.04059578571468592, "clip_ratio/high_max": 0.04059578571468592, "clip_ratio/region_mean": 0.09543395042419434, "reward_total_mean": 0.6184622049331665, "reward_meter_mean": 0.6009846925735474, "reward_meter_std": 0.31950291991233826, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8851910829544067, "reward_repeat_soft_std": 0.056723177433013916, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.6184622049331665, "reward_total_composite_std": 0.1360778957605362} {"timestamp_utc": "2026-04-13T01:56:28Z", "mode": "train", "global_step": 1450, "epoch": 0.14565544952285284, "loss": 0.0314, "grad_norm": 12.801274299621582, "learning_rate": 5.60909090909091e-06, "num_tokens": 2601554.0, "completions/mean_length": 49.875, "completions/min_length": 37.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.875, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9913651347160339, "rewards/meter/std": 0.004485689103603363, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9603877067565918, "rewards/repeat_soft/std": 0.022364236414432526, "rewards/judge_quality/mean": 0.32124999165534973, "rewards/judge_quality/std": 0.09876921027898788, "rewards/total_composite/mean": 0.779153048992157, "rewards/total_composite/std": 0.04693257063627243, "reward": 0.779153048992157, "reward_std": 0.04693257063627243, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17986398935317993, "sampling/sampling_logp_difference/max": 1.6085433959960938, "sampling/importance_sampling_ratio/min": 0.20017899572849274, "sampling/importance_sampling_ratio/mean": 1.0148694515228271, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3512857630848885, "clip_ratio/low_mean": 0.09481954760849476, "clip_ratio/low_min": 0.09481954760849476, "clip_ratio/high_mean": 0.05187665391713381, "clip_ratio/high_max": 0.05187665391713381, "clip_ratio/region_mean": 0.14669620152562857, "reward_total_mean": 0.779153048992157, "reward_meter_mean": 0.9913651347160339, "reward_meter_std": 0.004485689103603363, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9603877067565918, "reward_repeat_soft_std": 0.022364236414432526, "reward_judge_quality_mean": 0.32124999165534973, "reward_judge_quality_std": 0.09876921027898788, "reward_total_composite_mean": 0.779153048992157, "reward_total_composite_std": 0.04693257063627243} {"timestamp_utc": "2026-04-13T01:57:20Z", "mode": "eval", "global_step": 1450, "epoch": 0.14565544952285284, "eval_loss": NaN, "eval_runtime": 51.927, "eval_samples_per_second": 1.541, "eval_steps_per_second": 0.193, "eval_num_tokens": 2601554.0, "eval_completions/mean_length": 84.425, "eval_completions/min_length": 32.8, "eval_completions/max_length": 200.6, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 73.12321510314942, "eval_completions/min_terminated_length": 32.8, "eval_completions/max_terminated_length": 119.3, "eval_rewards/meter/mean": 0.9054690003395081, "eval_rewards/meter/std": 0.1729800846427679, "eval_rewards/count_adherence/mean": 0.9839583277702332, "eval_rewards/count_adherence/std": 0.0396198745816946, "eval_rewards/hard_gate/mean": 0.9875, "eval_rewards/hard_gate/std": 0.03535533845424652, "eval_rewards/repeat_soft/mean": 0.8351473569869995, "eval_rewards/repeat_soft/std": 0.14027236588299274, "eval_rewards/judge_quality/mean": 0.3716249942779541, "eval_rewards/judge_quality/std": 0.1617813564836979, "eval_rewards/total_composite/mean": 0.7440641462802887, "eval_rewards/total_composite/std": 0.11278714407235384, "eval_reward": 0.7440641462802887, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.0610603678971529, "eval_sampling/sampling_logp_difference/max": 1.0160212516784668, "eval_sampling/importance_sampling_ratio/min": 0.37212527096271514, "eval_sampling/importance_sampling_ratio/mean": 1.0140317678451538, "eval_sampling/importance_sampling_ratio/max": 1.5052916049957275, "eval_entropy": 0.6713622748851776, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7440641462802887, "eval_reward_meter_mean": 0.9054690003395081, "eval_reward_meter_std": 0.1729800846427679, "eval_reward_count_adherence_mean": 0.9839583277702332, "eval_reward_count_adherence_std": 0.0396198745816946, "eval_reward_hard_gate_mean": 0.9875, "eval_reward_hard_gate_std": 0.03535533845424652, "eval_reward_repeat_soft_mean": 0.8351473569869995, "eval_reward_repeat_soft_std": 0.14027236588299274, "eval_reward_judge_quality_mean": 0.3716249942779541, "eval_reward_judge_quality_std": 0.1617813564836979, "eval_reward_total_composite_mean": 0.7440641462802887, "eval_reward_total_composite_std": 0.11278714407235384} {"timestamp_utc": "2026-04-13T01:57:29Z", "mode": "train", "global_step": 1451, "epoch": 0.14575590155700652, "loss": 0.0385, "grad_norm": 19.684307098388672, "learning_rate": 5.606060606060606e-06, "num_tokens": 2603030.0, "completions/mean_length": 31.5, "completions/min_length": 26.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.5, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.8498288989067078, "rewards/meter/std": 0.24340075254440308, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9880582094192505, "rewards/repeat_soft/std": 0.020225683227181435, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.1940544992685318, "rewards/total_composite/mean": 0.7707288265228271, "rewards/total_composite/std": 0.12732051312923431, "reward": 0.7707288265228271, "reward_std": 0.12732049822807312, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13327187299728394, "sampling/sampling_logp_difference/max": 1.8691463470458984, "sampling/importance_sampling_ratio/min": 0.15425528585910797, "sampling/importance_sampling_ratio/mean": 1.0212337970733643, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0438485443592072, "clip_ratio/low_mean": 0.03098445199429989, "clip_ratio/low_min": 0.03098445199429989, "clip_ratio/high_mean": 0.09289148449897766, "clip_ratio/high_max": 0.09289148449897766, "clip_ratio/region_mean": 0.12387593649327755, "reward_total_mean": 0.7707288265228271, "reward_meter_mean": 0.8498288989067078, "reward_meter_std": 0.24340075254440308, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9880582094192505, "reward_repeat_soft_std": 0.020225683227181435, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.1940544992685318, "reward_total_composite_mean": 0.7707288265228271, "reward_total_composite_std": 0.12732051312923431} {"timestamp_utc": "2026-04-13T01:57:36Z", "mode": "train", "global_step": 1452, "epoch": 0.14585635359116023, "loss": 0.0364, "grad_norm": 13.280774116516113, "learning_rate": 5.603030303030303e-06, "num_tokens": 2604403.0, "completions/mean_length": 23.625, "completions/min_length": 22.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.625, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.9829362034797668, "rewards/meter/std": 0.012043765746057034, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9602519273757935, "rewards/repeat_soft/std": 0.006358357612043619, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.8728464841842651, "rewards/total_composite/std": 0.07865109294652939, "reward": 0.8728464841842651, "reward_std": 0.0786510780453682, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06365568190813065, "sampling/sampling_logp_difference/max": 0.8168649673461914, "sampling/importance_sampling_ratio/min": 0.4418145716190338, "sampling/importance_sampling_ratio/mean": 1.0108627080917358, "sampling/importance_sampling_ratio/max": 1.9689276218414307, "entropy": 0.4220782741904259, "clip_ratio/low_mean": 0.02641380438581109, "clip_ratio/low_min": 0.02641380438581109, "clip_ratio/high_mean": 0.044013505801558495, "clip_ratio/high_max": 0.044013505801558495, "clip_ratio/region_mean": 0.07042731018736959, "reward_total_mean": 0.8728464841842651, "reward_meter_mean": 0.9829362034797668, "reward_meter_std": 0.012043765746057034, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9602519273757935, "reward_repeat_soft_std": 0.006358357612043619, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.8728464841842651, "reward_total_composite_std": 0.07865109294652939} {"timestamp_utc": "2026-04-13T01:57:42Z", "mode": "train", "global_step": 1453, "epoch": 0.1459568056253139, "loss": -0.0206, "grad_norm": 9.184662818908691, "learning_rate": 5.600000000000001e-06, "num_tokens": 2605985.0, "completions/mean_length": 41.75, "completions/min_length": 39.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.75, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.9704310297966003, "rewards/meter/std": 0.03263917565345764, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8190011382102966, "rewards/repeat_soft/std": 0.10245952010154724, "rewards/judge_quality/mean": 0.30124998092651367, "rewards/judge_quality/std": 0.10398317128419876, "rewards/total_composite/mean": 0.7589690685272217, "rewards/total_composite/std": 0.03222963958978653, "reward": 0.7589690685272217, "reward_std": 0.03222961351275444, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10120944678783417, "sampling/sampling_logp_difference/max": 1.3301563262939453, "sampling/importance_sampling_ratio/min": 0.26443591713905334, "sampling/importance_sampling_ratio/mean": 1.017798900604248, "sampling/importance_sampling_ratio/max": 1.760914921760559, "entropy": 0.6464195884764194, "clip_ratio/low_mean": 0.0370616689324379, "clip_ratio/low_min": 0.0370616689324379, "clip_ratio/high_mean": 0.06864246726036072, "clip_ratio/high_max": 0.06864246726036072, "clip_ratio/region_mean": 0.10570413619279861, "reward_total_mean": 0.7589690685272217, "reward_meter_mean": 0.9704310297966003, "reward_meter_std": 0.03263917565345764, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8190011382102966, "reward_repeat_soft_std": 0.10245952010154724, "reward_judge_quality_mean": 0.30124998092651367, "reward_judge_quality_std": 0.10398317128419876, "reward_total_composite_mean": 0.7589690685272217, "reward_total_composite_std": 0.03222963958978653} {"timestamp_utc": "2026-04-13T01:57:51Z", "mode": "train", "global_step": 1454, "epoch": 0.14605725765946762, "loss": 0.0081, "grad_norm": 10.257253646850586, "learning_rate": 5.596969696969697e-06, "num_tokens": 2608146.0, "completions/mean_length": 87.125, "completions/min_length": 83.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.125, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.9765577912330627, "rewards/meter/std": 0.0396452359855175, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.556064248085022, "rewards/repeat_soft/std": 0.1769542694091797, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.2327437400817871, "rewards/total_composite/mean": 0.7669324278831482, "rewards/total_composite/std": 0.06753988564014435, "reward": 0.7669324278831482, "reward_std": 0.06753988564014435, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07204115390777588, "sampling/sampling_logp_difference/max": 1.6894121170043945, "sampling/importance_sampling_ratio/min": 0.18462802469730377, "sampling/importance_sampling_ratio/mean": 1.0013446807861328, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3411828391253948, "clip_ratio/low_mean": 0.01523442950565368, "clip_ratio/low_min": 0.01523442950565368, "clip_ratio/high_mean": 0.056852410081773996, "clip_ratio/high_max": 0.056852410081773996, "clip_ratio/region_mean": 0.07208683958742768, "reward_total_mean": 0.7669324278831482, "reward_meter_mean": 0.9765577912330627, "reward_meter_std": 0.0396452359855175, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.556064248085022, "reward_repeat_soft_std": 0.1769542694091797, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.2327437400817871, "reward_total_composite_mean": 0.7669324278831482, "reward_total_composite_std": 0.06753988564014435} {"timestamp_utc": "2026-04-13T01:57:58Z", "mode": "train", "global_step": 1455, "epoch": 0.1461577096936213, "loss": 0.0076, "grad_norm": 8.783917427062988, "learning_rate": 5.593939393939395e-06, "num_tokens": 2609827.0, "completions/mean_length": 47.125, "completions/min_length": 44.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.125, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.988781213760376, "rewards/meter/std": 0.006235369481146336, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9382025003433228, "rewards/repeat_soft/std": 0.04244663566350937, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.1865811049938202, "rewards/total_composite/mean": 0.8481467962265015, "rewards/total_composite/std": 0.05560413748025894, "reward": 0.8481467962265015, "reward_std": 0.05560414493083954, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11105208098888397, "sampling/sampling_logp_difference/max": 2.332024335861206, "sampling/importance_sampling_ratio/min": 0.25236275792121887, "sampling/importance_sampling_ratio/mean": 1.0206798315048218, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7171991392970085, "clip_ratio/low_mean": 0.09175567794591188, "clip_ratio/low_min": 0.09175567794591188, "clip_ratio/high_mean": 0.03630434721708298, "clip_ratio/high_max": 0.03630434721708298, "clip_ratio/region_mean": 0.12806002516299486, "reward_total_mean": 0.8481467962265015, "reward_meter_mean": 0.988781213760376, "reward_meter_std": 0.006235369481146336, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9382025003433228, "reward_repeat_soft_std": 0.04244663566350937, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.1865811049938202, "reward_total_composite_mean": 0.8481467962265015, "reward_total_composite_std": 0.05560413748025894} {"timestamp_utc": "2026-04-13T01:58:05Z", "mode": "train", "global_step": 1456, "epoch": 0.14625816172777498, "loss": 0.2124, "grad_norm": 11.923067092895508, "learning_rate": 5.5909090909090915e-06, "num_tokens": 2611496.0, "completions/mean_length": 50.625, "completions/min_length": 37.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.625, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.8726128339767456, "rewards/meter/std": 0.20927084982395172, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9375370740890503, "rewards/repeat_soft/std": 0.03802049160003662, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7373044490814209, "rewards/total_composite/std": 0.11701110750436783, "reward": 0.7373044490814209, "reward_std": 0.11701110005378723, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13205182552337646, "sampling/sampling_logp_difference/max": 2.409846305847168, "sampling/importance_sampling_ratio/min": 0.08982910215854645, "sampling/importance_sampling_ratio/mean": 1.0029274225234985, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.836461529135704, "clip_ratio/low_mean": 0.021939328871667385, "clip_ratio/low_min": 0.021939328871667385, "clip_ratio/high_mean": 0.10610067751258612, "clip_ratio/high_max": 0.10610067751258612, "clip_ratio/region_mean": 0.1280400063842535, "reward_total_mean": 0.7373044490814209, "reward_meter_mean": 0.8726128339767456, "reward_meter_std": 0.20927084982395172, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9375370740890503, "reward_repeat_soft_std": 0.03802049160003662, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7373044490814209, "reward_total_composite_std": 0.11701110750436783} {"timestamp_utc": "2026-04-13T01:58:13Z", "mode": "train", "global_step": 1457, "epoch": 0.14635861376192869, "loss": 0.0137, "grad_norm": 15.671184539794922, "learning_rate": 5.587878787878789e-06, "num_tokens": 2613594.0, "completions/mean_length": 83.25, "completions/min_length": 78.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.25, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.9229756593704224, "rewards/meter/std": 0.19842661917209625, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6378737688064575, "rewards/repeat_soft/std": 0.08311189711093903, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.12631450593471527, "rewards/total_composite/mean": 0.7285014390945435, "rewards/total_composite/std": 0.11893574893474579, "reward": 0.7285014390945435, "reward_std": 0.11893574893474579, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06636929512023926, "sampling/sampling_logp_difference/max": 1.302107572555542, "sampling/importance_sampling_ratio/min": 0.27195802330970764, "sampling/importance_sampling_ratio/mean": 1.011521339416504, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4298529140651226, "clip_ratio/low_mean": 0.018672839738428593, "clip_ratio/low_min": 0.018672839738428593, "clip_ratio/high_mean": 0.05187793681398034, "clip_ratio/high_max": 0.05187793681398034, "clip_ratio/region_mean": 0.07055077655240893, "reward_total_mean": 0.7285014390945435, "reward_meter_mean": 0.9229756593704224, "reward_meter_std": 0.19842661917209625, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6378737688064575, "reward_repeat_soft_std": 0.08311189711093903, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.12631450593471527, "reward_total_composite_mean": 0.7285014390945435, "reward_total_composite_std": 0.11893574893474579} {"timestamp_utc": "2026-04-13T01:58:20Z", "mode": "train", "global_step": 1458, "epoch": 0.14645906579608237, "loss": 0.0029, "grad_norm": 8.01706600189209, "learning_rate": 5.584848484848485e-06, "num_tokens": 2615970.0, "completions/mean_length": 95.0, "completions/min_length": 81.0, "completions/max_length": 113.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.0, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.8610199689865112, "rewards/meter/std": 0.21744556725025177, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8615142107009888, "rewards/repeat_soft/std": 0.1323707550764084, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.1249857097864151, "rewards/total_composite/mean": 0.7293603420257568, "rewards/total_composite/std": 0.0927034392952919, "reward": 0.7293603420257568, "reward_std": 0.0927034392952919, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11957994103431702, "sampling/sampling_logp_difference/max": 2.0020008087158203, "sampling/importance_sampling_ratio/min": 0.13506478071212769, "sampling/importance_sampling_ratio/mean": 1.0241739749908447, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8755145184695721, "clip_ratio/low_mean": 0.056615530513226986, "clip_ratio/low_min": 0.056615530513226986, "clip_ratio/high_mean": 0.05061599612236023, "clip_ratio/high_max": 0.05061599612236023, "clip_ratio/region_mean": 0.10723152663558722, "reward_total_mean": 0.7293603420257568, "reward_meter_mean": 0.8610199689865112, "reward_meter_std": 0.21744556725025177, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8615142107009888, "reward_repeat_soft_std": 0.1323707550764084, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.1249857097864151, "reward_total_composite_mean": 0.7293603420257568, "reward_total_composite_std": 0.0927034392952919} {"timestamp_utc": "2026-04-13T01:58:26Z", "mode": "train", "global_step": 1459, "epoch": 0.14655951783023607, "loss": 0.0238, "grad_norm": 15.472408294677734, "learning_rate": 5.5818181818181824e-06, "num_tokens": 2617745.0, "completions/mean_length": 46.875, "completions/min_length": 44.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.875, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9865671396255493, "rewards/meter/std": 0.015957901254296303, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.98426353931427, "rewards/repeat_soft/std": 0.022312788292765617, "rewards/judge_quality/mean": 0.44624999165534973, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8262565732002258, "rewards/total_composite/std": 0.006980689242482185, "reward": 0.8262565732002258, "reward_std": 0.006980689242482185, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11751192063093185, "sampling/sampling_logp_difference/max": 1.2802505493164062, "sampling/importance_sampling_ratio/min": 0.277967631816864, "sampling/importance_sampling_ratio/mean": 1.0117177963256836, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8609643802046776, "clip_ratio/low_mean": 0.04748217482119799, "clip_ratio/low_min": 0.04748217482119799, "clip_ratio/high_mean": 0.07719038426876068, "clip_ratio/high_max": 0.07719038426876068, "clip_ratio/region_mean": 0.12467255908995867, "reward_total_mean": 0.8262565732002258, "reward_meter_mean": 0.9865671396255493, "reward_meter_std": 0.015957901254296303, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.98426353931427, "reward_repeat_soft_std": 0.022312788292765617, "reward_judge_quality_mean": 0.44624999165534973, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8262565732002258, "reward_total_composite_std": 0.006980689242482185} {"timestamp_utc": "2026-04-13T01:58:34Z", "mode": "train", "global_step": 1460, "epoch": 0.14665996986438976, "loss": 0.0269, "grad_norm": 10.019533157348633, "learning_rate": 5.578787878787879e-06, "num_tokens": 2620034.0, "completions/mean_length": 109.125, "completions/min_length": 98.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.125, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.3033667206764221, "rewards/meter/std": 0.3250879645347595, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8303238153457642, "rewards/repeat_soft/std": 0.08837782591581345, "rewards/judge_quality/mean": 0.3725000023841858, "rewards/judge_quality/std": 0.16368524730205536, "rewards/total_composite/mean": 0.4812973737716675, "rewards/total_composite/std": 0.14519982039928436, "reward": 0.4812973737716675, "reward_std": 0.14519980549812317, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12545554339885712, "sampling/sampling_logp_difference/max": 5.262222766876221, "sampling/importance_sampling_ratio/min": 0.005183769855648279, "sampling/importance_sampling_ratio/mean": 1.0054506063461304, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.525087520480156, "clip_ratio/low_mean": 0.054165286011993885, "clip_ratio/low_min": 0.054165286011993885, "clip_ratio/high_mean": 0.06152481399476528, "clip_ratio/high_max": 0.06152481399476528, "clip_ratio/region_mean": 0.11569010000675917, "reward_total_mean": 0.4812973737716675, "reward_meter_mean": 0.3033667206764221, "reward_meter_std": 0.3250879645347595, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8303238153457642, "reward_repeat_soft_std": 0.08837782591581345, "reward_judge_quality_mean": 0.3725000023841858, "reward_judge_quality_std": 0.16368524730205536, "reward_total_composite_mean": 0.4812973737716675, "reward_total_composite_std": 0.14519982039928436} {"timestamp_utc": "2026-04-13T01:58:41Z", "mode": "train", "global_step": 1461, "epoch": 0.14676042189854344, "loss": -0.0162, "grad_norm": 8.631481170654297, "learning_rate": 5.575757575757577e-06, "num_tokens": 2621994.0, "completions/mean_length": 66.0, "completions/min_length": 59.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.9836721420288086, "rewards/meter/std": 0.005598485004156828, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8110396265983582, "rewards/repeat_soft/std": 0.10252104699611664, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.7742564678192139, "rewards/total_composite/std": 0.03209323808550835, "reward": 0.7742564678192139, "reward_std": 0.03209323808550835, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08618605881929398, "sampling/sampling_logp_difference/max": 1.3737215995788574, "sampling/importance_sampling_ratio/min": 0.25316303968429565, "sampling/importance_sampling_ratio/mean": 1.0104362964630127, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.566940613090992, "clip_ratio/low_mean": 0.04584355093538761, "clip_ratio/low_min": 0.04584355093538761, "clip_ratio/high_mean": 0.06134584825485945, "clip_ratio/high_max": 0.06134584825485945, "clip_ratio/region_mean": 0.10718939919024706, "reward_total_mean": 0.7742564678192139, "reward_meter_mean": 0.9836721420288086, "reward_meter_std": 0.005598485004156828, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8110396265983582, "reward_repeat_soft_std": 0.10252104699611664, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.7742564678192139, "reward_total_composite_std": 0.03209323808550835} {"timestamp_utc": "2026-04-13T01:58:48Z", "mode": "train", "global_step": 1462, "epoch": 0.14686087393269714, "loss": 0.0491, "grad_norm": 10.123976707458496, "learning_rate": 5.572727272727273e-06, "num_tokens": 2623850.0, "completions/mean_length": 76.0, "completions/min_length": 65.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.0, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.7389377355575562, "rewards/meter/std": 0.25719138979911804, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9125741124153137, "rewards/repeat_soft/std": 0.062396734952926636, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.6806544065475464, "rewards/total_composite/std": 0.10745598375797272, "reward": 0.6806544065475464, "reward_std": 0.10745598375797272, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12425123155117035, "sampling/sampling_logp_difference/max": 1.8901863098144531, "sampling/importance_sampling_ratio/min": 0.15104366838932037, "sampling/importance_sampling_ratio/mean": 1.0096882581710815, "sampling/importance_sampling_ratio/max": 1.974582314491272, "entropy": 0.9619928784668446, "clip_ratio/low_mean": 0.05356930289417505, "clip_ratio/low_min": 0.05356930289417505, "clip_ratio/high_mean": 0.06757452338933945, "clip_ratio/high_max": 0.06757452338933945, "clip_ratio/region_mean": 0.1211438262835145, "reward_total_mean": 0.6806544065475464, "reward_meter_mean": 0.7389377355575562, "reward_meter_std": 0.25719138979911804, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9125741124153137, "reward_repeat_soft_std": 0.062396734952926636, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.6806544065475464, "reward_total_composite_std": 0.10745598375797272} {"timestamp_utc": "2026-04-13T01:58:56Z", "mode": "train", "global_step": 1463, "epoch": 0.14696132596685083, "loss": 0.1089, "grad_norm": 23.043766021728516, "learning_rate": 5.569696969696971e-06, "num_tokens": 2625423.0, "completions/mean_length": 26.625, "completions/min_length": 24.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.625, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.8817969560623169, "rewards/meter/std": 0.20345740020275116, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9565733671188354, "rewards/repeat_soft/std": 0.013593671843409538, "rewards/judge_quality/mean": 0.38749998807907104, "rewards/judge_quality/std": 0.11877349019050598, "rewards/total_composite/mean": 0.7587159872055054, "rewards/total_composite/std": 0.08736684173345566, "reward": 0.7587159872055054, "reward_std": 0.08736684918403625, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13849620521068573, "sampling/sampling_logp_difference/max": 1.4168167114257812, "sampling/importance_sampling_ratio/min": 0.2424846887588501, "sampling/importance_sampling_ratio/mean": 1.0355103015899658, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0917094945907593, "clip_ratio/low_mean": 0.0496241832152009, "clip_ratio/low_min": 0.0496241832152009, "clip_ratio/high_mean": 0.0922186616808176, "clip_ratio/high_max": 0.0922186616808176, "clip_ratio/region_mean": 0.1418428448960185, "reward_total_mean": 0.7587159872055054, "reward_meter_mean": 0.8817969560623169, "reward_meter_std": 0.20345740020275116, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9565733671188354, "reward_repeat_soft_std": 0.013593671843409538, "reward_judge_quality_mean": 0.38749998807907104, "reward_judge_quality_std": 0.11877349019050598, "reward_total_composite_mean": 0.7587159872055054, "reward_total_composite_std": 0.08736684173345566} {"timestamp_utc": "2026-04-13T01:59:03Z", "mode": "train", "global_step": 1464, "epoch": 0.14706177800100453, "loss": 0.0083, "grad_norm": 13.523809432983398, "learning_rate": 5.566666666666667e-06, "num_tokens": 2627110.0, "completions/mean_length": 46.875, "completions/min_length": 43.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.875, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9888760447502136, "rewards/meter/std": 0.0076562934555113316, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9197086691856384, "rewards/repeat_soft/std": 0.07127145677804947, "rewards/judge_quality/mean": 0.4437499940395355, "rewards/judge_quality/std": 0.2084595113992691, "rewards/total_composite/mean": 0.8200900554656982, "rewards/total_composite/std": 0.06687629967927933, "reward": 0.8200900554656982, "reward_std": 0.06687628477811813, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10359450429677963, "sampling/sampling_logp_difference/max": 0.946136474609375, "sampling/importance_sampling_ratio/min": 0.3882381021976471, "sampling/importance_sampling_ratio/mean": 0.9969365000724792, "sampling/importance_sampling_ratio/max": 1.7342573404312134, "entropy": 0.7142519615590572, "clip_ratio/low_mean": 0.033877977170050144, "clip_ratio/low_min": 0.033877977170050144, "clip_ratio/high_mean": 0.08261867891997099, "clip_ratio/high_max": 0.08261867891997099, "clip_ratio/region_mean": 0.11649665609002113, "reward_total_mean": 0.8200900554656982, "reward_meter_mean": 0.9888760447502136, "reward_meter_std": 0.0076562934555113316, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9197086691856384, "reward_repeat_soft_std": 0.07127145677804947, "reward_judge_quality_mean": 0.4437499940395355, "reward_judge_quality_std": 0.2084595113992691, "reward_total_composite_mean": 0.8200900554656982, "reward_total_composite_std": 0.06687629967927933} {"timestamp_utc": "2026-04-13T01:59:20Z", "mode": "train", "global_step": 1465, "epoch": 0.14716223003515821, "loss": -0.1361, "grad_norm": 1.7550452947616577, "learning_rate": 5.563636363636364e-06, "num_tokens": 2628639.0, "completions/mean_length": 110.125, "completions/min_length": 46.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 52.71428680419922, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.8632482886314392, "rewards/meter/std": 0.3491761386394501, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.3720119297504425, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9212003946304321, "rewards/repeat_soft/std": 0.08390890061855316, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.13274572789669037, "rewards/total_composite/mean": 0.7013317346572876, "rewards/total_composite/std": 0.28527340292930603, "reward": 0.7013317346572876, "reward_std": 0.28527340292930603, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12262234091758728, "sampling/sampling_logp_difference/max": 1.4254236221313477, "sampling/importance_sampling_ratio/min": 0.24040661752223969, "sampling/importance_sampling_ratio/mean": 1.0058722496032715, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6241954416036606, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.1099180867895484, "clip_ratio/high_max": 0.1099180867895484, "clip_ratio/region_mean": 0.1099180867895484, "reward_total_mean": 0.7013317346572876, "reward_meter_mean": 0.8632482886314392, "reward_meter_std": 0.3491761386394501, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.3720119297504425, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9212003946304321, "reward_repeat_soft_std": 0.08390890061855316, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.13274572789669037, "reward_total_composite_mean": 0.7013317346572876, "reward_total_composite_std": 0.28527340292930603} {"timestamp_utc": "2026-04-13T01:59:26Z", "mode": "train", "global_step": 1466, "epoch": 0.1472626820693119, "loss": -0.0288, "grad_norm": 12.002464294433594, "learning_rate": 5.560606060606061e-06, "num_tokens": 2630154.0, "completions/mean_length": 46.375, "completions/min_length": 43.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.375, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9778307676315308, "rewards/meter/std": 0.026063693687319756, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8372169137001038, "rewards/repeat_soft/std": 0.0865527018904686, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.7944954633712769, "rewards/total_composite/std": 0.022956350818276405, "reward": 0.7944954633712769, "reward_std": 0.022956348955631256, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1009690910577774, "sampling/sampling_logp_difference/max": 1.321065902709961, "sampling/importance_sampling_ratio/min": 0.307416170835495, "sampling/importance_sampling_ratio/mean": 1.0018151998519897, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4906904138624668, "clip_ratio/low_mean": 0.03926767781376839, "clip_ratio/low_min": 0.03926767781376839, "clip_ratio/high_mean": 0.08041759137995541, "clip_ratio/high_max": 0.08041759137995541, "clip_ratio/region_mean": 0.1196852691937238, "reward_total_mean": 0.7944954633712769, "reward_meter_mean": 0.9778307676315308, "reward_meter_std": 0.026063693687319756, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8372169137001038, "reward_repeat_soft_std": 0.0865527018904686, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.7944954633712769, "reward_total_composite_std": 0.022956350818276405} {"timestamp_utc": "2026-04-13T01:59:34Z", "mode": "train", "global_step": 1467, "epoch": 0.1473631341034656, "loss": 0.027, "grad_norm": 9.719688415527344, "learning_rate": 5.557575757575758e-06, "num_tokens": 2632106.0, "completions/mean_length": 72.0, "completions/min_length": 61.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.9903452396392822, "rewards/meter/std": 0.007049769628793001, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9019270539283752, "rewards/repeat_soft/std": 0.06449790298938751, "rewards/judge_quality/mean": 0.41874998807907104, "rewards/judge_quality/std": 0.14574319124221802, "rewards/total_composite/mean": 0.8114730715751648, "rewards/total_composite/std": 0.046837013214826584, "reward": 0.8114730715751648, "reward_std": 0.04683702811598778, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11077554523944855, "sampling/sampling_logp_difference/max": 1.5591907501220703, "sampling/importance_sampling_ratio/min": 0.21030618250370026, "sampling/importance_sampling_ratio/mean": 1.008925437927246, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8807422667741776, "clip_ratio/low_mean": 0.05454443255439401, "clip_ratio/low_min": 0.05454443255439401, "clip_ratio/high_mean": 0.046446639113128185, "clip_ratio/high_max": 0.046446639113128185, "clip_ratio/region_mean": 0.10099107166752219, "reward_total_mean": 0.8114730715751648, "reward_meter_mean": 0.9903452396392822, "reward_meter_std": 0.007049769628793001, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9019270539283752, "reward_repeat_soft_std": 0.06449790298938751, "reward_judge_quality_mean": 0.41874998807907104, "reward_judge_quality_std": 0.14574319124221802, "reward_total_composite_mean": 0.8114730715751648, "reward_total_composite_std": 0.046837013214826584} {"timestamp_utc": "2026-04-13T01:59:40Z", "mode": "train", "global_step": 1468, "epoch": 0.14746358613761928, "loss": 0.0454, "grad_norm": 16.11932373046875, "learning_rate": 5.554545454545454e-06, "num_tokens": 2633427.0, "completions/mean_length": 28.125, "completions/min_length": 18.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.125, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.8573204278945923, "rewards/meter/std": 0.33501383662223816, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.95582115650177, "rewards/repeat_soft/std": 0.011827274225652218, "rewards/judge_quality/mean": 0.3962499797344208, "rewards/judge_quality/std": 0.09085899591445923, "rewards/total_composite/mean": 0.750251293182373, "rewards/total_composite/std": 0.14695875346660614, "reward": 0.750251293182373, "reward_std": 0.14695876836776733, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16055671870708466, "sampling/sampling_logp_difference/max": 2.0277528762817383, "sampling/importance_sampling_ratio/min": 0.13163097202777863, "sampling/importance_sampling_ratio/mean": 0.9869988560676575, "sampling/importance_sampling_ratio/max": 1.8777732849121094, "entropy": 1.217717632651329, "clip_ratio/low_mean": 0.0223214291036129, "clip_ratio/low_min": 0.0223214291036129, "clip_ratio/high_mean": 0.14957229420542717, "clip_ratio/high_max": 0.14957229420542717, "clip_ratio/region_mean": 0.17189372330904007, "reward_total_mean": 0.750251293182373, "reward_meter_mean": 0.8573204278945923, "reward_meter_std": 0.33501383662223816, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.95582115650177, "reward_repeat_soft_std": 0.011827274225652218, "reward_judge_quality_mean": 0.3962499797344208, "reward_judge_quality_std": 0.09085899591445923, "reward_total_composite_mean": 0.750251293182373, "reward_total_composite_std": 0.14695875346660614} {"timestamp_utc": "2026-04-13T01:59:48Z", "mode": "train", "global_step": 1469, "epoch": 0.14756403817177297, "loss": 0.0171, "grad_norm": 6.1086273193359375, "learning_rate": 5.5515151515151524e-06, "num_tokens": 2636053.0, "completions/mean_length": 151.25, "completions/min_length": 136.0, "completions/max_length": 164.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 151.25, "completions/min_terminated_length": 136.0, "completions/max_terminated_length": 164.0, "rewards/meter/mean": 0.8527553081512451, "rewards/meter/std": 0.18139947950839996, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6117408275604248, "rewards/repeat_soft/std": 0.11345726251602173, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.6916639804840088, "rewards/total_composite/std": 0.08975062519311905, "reward": 0.6916639804840088, "reward_std": 0.08975062519311905, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0887717604637146, "sampling/sampling_logp_difference/max": 2.793405532836914, "sampling/importance_sampling_ratio/min": 0.06121239438652992, "sampling/importance_sampling_ratio/mean": 1.0098897218704224, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47371668368577957, "clip_ratio/low_mean": 0.030146121978759766, "clip_ratio/low_min": 0.030146121978759766, "clip_ratio/high_mean": 0.049893836956471205, "clip_ratio/high_max": 0.049893836956471205, "clip_ratio/region_mean": 0.08003995893523097, "reward_total_mean": 0.6916639804840088, "reward_meter_mean": 0.8527553081512451, "reward_meter_std": 0.18139947950839996, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6117408275604248, "reward_repeat_soft_std": 0.11345726251602173, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.6916639804840088, "reward_total_composite_std": 0.08975062519311905} {"timestamp_utc": "2026-04-13T01:59:56Z", "mode": "train", "global_step": 1470, "epoch": 0.14766449020592667, "loss": 0.0109, "grad_norm": 6.409454822540283, "learning_rate": 5.548484848484849e-06, "num_tokens": 2638720.0, "completions/mean_length": 138.375, "completions/min_length": 123.0, "completions/max_length": 149.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 138.375, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 149.0, "rewards/meter/mean": 0.9869641661643982, "rewards/meter/std": 0.0050875903107225895, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8770157098770142, "rewards/repeat_soft/std": 0.059511564671993256, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7887104749679565, "rewards/total_composite/std": 0.026418399065732956, "reward": 0.7887104749679565, "reward_std": 0.02641838975250721, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11286922544240952, "sampling/sampling_logp_difference/max": 2.4517292976379395, "sampling/importance_sampling_ratio/min": 0.08614448457956314, "sampling/importance_sampling_ratio/mean": 1.0247235298156738, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9983518049120903, "clip_ratio/low_mean": 0.04846068471670151, "clip_ratio/low_min": 0.04846068471670151, "clip_ratio/high_mean": 0.06431647762656212, "clip_ratio/high_max": 0.06431647762656212, "clip_ratio/region_mean": 0.11277716234326363, "reward_total_mean": 0.7887104749679565, "reward_meter_mean": 0.9869641661643982, "reward_meter_std": 0.0050875903107225895, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8770157098770142, "reward_repeat_soft_std": 0.059511564671993256, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7887104749679565, "reward_total_composite_std": 0.026418399065732956} {"timestamp_utc": "2026-04-13T02:00:03Z", "mode": "train", "global_step": 1471, "epoch": 0.14776494224008035, "loss": 0.0441, "grad_norm": 10.04185962677002, "learning_rate": 5.545454545454546e-06, "num_tokens": 2640382.0, "completions/mean_length": 54.75, "completions/min_length": 51.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.75, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.97763991355896, "rewards/meter/std": 0.03211536630988121, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9007923603057861, "rewards/repeat_soft/std": 0.06607694923877716, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.8622671961784363, "rewards/total_composite/std": 0.08643404394388199, "reward": 0.8622671961784363, "reward_std": 0.08643403649330139, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1020558625459671, "sampling/sampling_logp_difference/max": 1.3350629806518555, "sampling/importance_sampling_ratio/min": 0.26314160227775574, "sampling/importance_sampling_ratio/mean": 1.0091861486434937, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7376570254564285, "clip_ratio/low_mean": 0.03836971428245306, "clip_ratio/low_min": 0.03836971428245306, "clip_ratio/high_mean": 0.06088768504559994, "clip_ratio/high_max": 0.06088768504559994, "clip_ratio/region_mean": 0.099257399328053, "reward_total_mean": 0.8622671961784363, "reward_meter_mean": 0.97763991355896, "reward_meter_std": 0.03211536630988121, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9007923603057861, "reward_repeat_soft_std": 0.06607694923877716, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.8622671961784363, "reward_total_composite_std": 0.08643404394388199} {"timestamp_utc": "2026-04-13T02:00:10Z", "mode": "train", "global_step": 1472, "epoch": 0.14786539427423406, "loss": 0.0458, "grad_norm": 12.927340507507324, "learning_rate": 5.5424242424242425e-06, "num_tokens": 2642111.0, "completions/mean_length": 52.125, "completions/min_length": 47.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.125, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9779243469238281, "rewards/meter/std": 0.015326487831771374, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9272279739379883, "rewards/repeat_soft/std": 0.05233953148126602, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.8057887554168701, "rewards/total_composite/std": 0.025771401822566986, "reward": 0.8057887554168701, "reward_std": 0.025771409273147583, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11424799263477325, "sampling/sampling_logp_difference/max": 3.053861141204834, "sampling/importance_sampling_ratio/min": 0.04717642068862915, "sampling/importance_sampling_ratio/mean": 1.019176959991455, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6815639697015285, "clip_ratio/low_mean": 0.027530016377568245, "clip_ratio/low_min": 0.027530016377568245, "clip_ratio/high_mean": 0.11175844352692366, "clip_ratio/high_max": 0.11175844352692366, "clip_ratio/region_mean": 0.1392884599044919, "reward_total_mean": 0.8057887554168701, "reward_meter_mean": 0.9779243469238281, "reward_meter_std": 0.015326487831771374, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9272279739379883, "reward_repeat_soft_std": 0.05233953148126602, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.8057887554168701, "reward_total_composite_std": 0.025771401822566986} {"timestamp_utc": "2026-04-13T02:00:18Z", "mode": "train", "global_step": 1473, "epoch": 0.14796584630838774, "loss": 0.0871, "grad_norm": 18.347198486328125, "learning_rate": 5.53939393939394e-06, "num_tokens": 2643694.0, "completions/mean_length": 39.875, "completions/min_length": 36.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.875, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.5484555959701538, "rewards/meter/std": 0.44192957878112793, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9592545032501221, "rewards/repeat_soft/std": 0.05122688412666321, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.6562304496765137, "rewards/total_composite/std": 0.15863797068595886, "reward": 0.6562304496765137, "reward_std": 0.15863795578479767, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13247160613536835, "sampling/sampling_logp_difference/max": 1.740262508392334, "sampling/importance_sampling_ratio/min": 0.17547433078289032, "sampling/importance_sampling_ratio/mean": 1.020572543144226, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7850005105137825, "clip_ratio/low_mean": 0.0807791817933321, "clip_ratio/low_min": 0.0807791817933321, "clip_ratio/high_mean": 0.07789321336895227, "clip_ratio/high_max": 0.07789321336895227, "clip_ratio/region_mean": 0.15867239516228437, "reward_total_mean": 0.6562304496765137, "reward_meter_mean": 0.5484555959701538, "reward_meter_std": 0.44192957878112793, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9592545032501221, "reward_repeat_soft_std": 0.05122688412666321, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.6562304496765137, "reward_total_composite_std": 0.15863797068595886} {"timestamp_utc": "2026-04-13T02:00:26Z", "mode": "train", "global_step": 1474, "epoch": 0.14806629834254142, "loss": 0.0353, "grad_norm": 10.08499813079834, "learning_rate": 5.536363636363636e-06, "num_tokens": 2645523.0, "completions/mean_length": 53.625, "completions/min_length": 48.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.625, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9894035458564758, "rewards/meter/std": 0.004185415338724852, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9736538529396057, "rewards/repeat_soft/std": 0.03134281933307648, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.8309720158576965, "rewards/total_composite/std": 0.033344656229019165, "reward": 0.8309720158576965, "reward_std": 0.033344656229019165, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13161441683769226, "sampling/sampling_logp_difference/max": 1.4556841850280762, "sampling/importance_sampling_ratio/min": 0.23324072360992432, "sampling/importance_sampling_ratio/mean": 1.0057982206344604, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0672649592161179, "clip_ratio/low_mean": 0.08844975661486387, "clip_ratio/low_min": 0.08844975661486387, "clip_ratio/high_mean": 0.04901960864663124, "clip_ratio/high_max": 0.04901960864663124, "clip_ratio/region_mean": 0.1374693652614951, "reward_total_mean": 0.8309720158576965, "reward_meter_mean": 0.9894035458564758, "reward_meter_std": 0.004185415338724852, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9736538529396057, "reward_repeat_soft_std": 0.03134281933307648, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.8309720158576965, "reward_total_composite_std": 0.033344656229019165} {"timestamp_utc": "2026-04-13T02:00:34Z", "mode": "train", "global_step": 1475, "epoch": 0.14816675037669513, "loss": 0.1197, "grad_norm": 10.141109466552734, "learning_rate": 5.533333333333334e-06, "num_tokens": 2648051.0, "completions/mean_length": 117.0, "completions/min_length": 93.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.0, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.8887219429016113, "rewards/meter/std": 0.16848894953727722, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9069532155990601, "rewards/repeat_soft/std": 0.08669080585241318, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7564952373504639, "rewards/total_composite/std": 0.0909142717719078, "reward": 0.7564952373504639, "reward_std": 0.09091425687074661, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1390935480594635, "sampling/sampling_logp_difference/max": 2.0390305519104004, "sampling/importance_sampling_ratio/min": 0.1301548331975937, "sampling/importance_sampling_ratio/mean": 0.9939952492713928, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8891730830073357, "clip_ratio/low_mean": 0.029493408277630806, "clip_ratio/low_min": 0.029493408277630806, "clip_ratio/high_mean": 0.11418080795556307, "clip_ratio/high_max": 0.11418080795556307, "clip_ratio/region_mean": 0.14367421623319387, "reward_total_mean": 0.7564952373504639, "reward_meter_mean": 0.8887219429016113, "reward_meter_std": 0.16848894953727722, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9069532155990601, "reward_repeat_soft_std": 0.08669080585241318, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7564952373504639, "reward_total_composite_std": 0.0909142717719078} {"timestamp_utc": "2026-04-13T02:00:48Z", "mode": "train", "global_step": 1476, "epoch": 0.14826720241084881, "loss": -0.0133, "grad_norm": 5.1464619636535645, "learning_rate": 5.530303030303031e-06, "num_tokens": 2649784.0, "completions/mean_length": 115.625, "completions/min_length": 47.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 59.000003814697266, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.6302094459533691, "rewards/meter/std": 0.3622470498085022, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.24800792336463928, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9451943635940552, "rewards/repeat_soft/std": 0.0727449581027031, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.13452960550785065, "rewards/total_composite/mean": 0.5618894100189209, "rewards/total_composite/std": 0.3124414384365082, "reward": 0.5618894100189209, "reward_std": 0.3124414086341858, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13134989142417908, "sampling/sampling_logp_difference/max": 1.6869049072265625, "sampling/importance_sampling_ratio/min": 0.1850915104150772, "sampling/importance_sampling_ratio/mean": 1.0072481632232666, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6992876976728439, "clip_ratio/low_mean": 0.030138544738292694, "clip_ratio/low_min": 0.030138544738292694, "clip_ratio/high_mean": 0.08890178240835667, "clip_ratio/high_max": 0.08890178240835667, "clip_ratio/region_mean": 0.11904032714664936, "reward_total_mean": 0.5618894100189209, "reward_meter_mean": 0.6302094459533691, "reward_meter_std": 0.3622470498085022, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.24800792336463928, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9451943635940552, "reward_repeat_soft_std": 0.0727449581027031, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.13452960550785065, "reward_total_composite_mean": 0.5618894100189209, "reward_total_composite_std": 0.3124414384365082} {"timestamp_utc": "2026-04-13T02:01:00Z", "mode": "train", "global_step": 1477, "epoch": 0.14836765444500252, "loss": 0.0247, "grad_norm": 7.303560256958008, "learning_rate": 5.527272727272728e-06, "num_tokens": 2653018.0, "completions/mean_length": 179.25, "completions/min_length": 168.0, "completions/max_length": 197.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 179.25, "completions/min_terminated_length": 168.0, "completions/max_terminated_length": 197.0, "rewards/meter/mean": 0.68544602394104, "rewards/meter/std": 0.3674994707107544, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7645697593688965, "rewards/repeat_soft/std": 0.1208936870098114, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.6482827067375183, "rewards/total_composite/std": 0.16852517426013947, "reward": 0.6482827067375183, "reward_std": 0.16852517426013947, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10492746531963348, "sampling/sampling_logp_difference/max": 1.7250514030456543, "sampling/importance_sampling_ratio/min": 0.17816390097141266, "sampling/importance_sampling_ratio/mean": 1.0127602815628052, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.733396016061306, "clip_ratio/low_mean": 0.030697056092321873, "clip_ratio/low_min": 0.030697056092321873, "clip_ratio/high_mean": 0.0631793150678277, "clip_ratio/high_max": 0.0631793150678277, "clip_ratio/region_mean": 0.09387637116014957, "reward_total_mean": 0.6482827067375183, "reward_meter_mean": 0.68544602394104, "reward_meter_std": 0.3674994707107544, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7645697593688965, "reward_repeat_soft_std": 0.1208936870098114, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.6482827067375183, "reward_total_composite_std": 0.16852517426013947} {"timestamp_utc": "2026-04-13T02:01:09Z", "mode": "train", "global_step": 1478, "epoch": 0.1484681064791562, "loss": 0.0109, "grad_norm": 7.672631740570068, "learning_rate": 5.524242424242424e-06, "num_tokens": 2655559.0, "completions/mean_length": 126.625, "completions/min_length": 104.0, "completions/max_length": 137.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.625, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.8789267539978027, "rewards/meter/std": 0.15898440778255463, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9223957061767578, "rewards/repeat_soft/std": 0.02840244211256504, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7573815584182739, "rewards/total_composite/std": 0.0891757383942604, "reward": 0.7573815584182739, "reward_std": 0.0891757383942604, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12265175580978394, "sampling/sampling_logp_difference/max": 2.356027364730835, "sampling/importance_sampling_ratio/min": 0.0947960689663887, "sampling/importance_sampling_ratio/mean": 1.0002391338348389, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.691898763179779, "clip_ratio/low_mean": 0.030437272042036057, "clip_ratio/low_min": 0.030437272042036057, "clip_ratio/high_mean": 0.07340242620557547, "clip_ratio/high_max": 0.07340242620557547, "clip_ratio/region_mean": 0.10383969824761152, "reward_total_mean": 0.7573815584182739, "reward_meter_mean": 0.8789267539978027, "reward_meter_std": 0.15898440778255463, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9223957061767578, "reward_repeat_soft_std": 0.02840244211256504, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7573815584182739, "reward_total_composite_std": 0.0891757383942604} {"timestamp_utc": "2026-04-13T02:01:22Z", "mode": "train", "global_step": 1479, "epoch": 0.14856855851330988, "loss": -0.1082, "grad_norm": 3.826810359954834, "learning_rate": 5.521212121212122e-06, "num_tokens": 2657079.0, "completions/mean_length": 107.0, "completions/min_length": 33.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 49.142860412597656, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.590712308883667, "rewards/meter/std": 0.4094976782798767, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9817673563957214, "rewards/repeat_soft/std": 0.026286648586392403, "rewards/judge_quality/mean": 0.39249998331069946, "rewards/judge_quality/std": 0.13905291259288788, "rewards/total_composite/mean": 0.5450050234794617, "rewards/total_composite/std": 0.2797129452228546, "reward": 0.5450050234794617, "reward_std": 0.2797129452228546, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13384504616260529, "sampling/sampling_logp_difference/max": 1.3628973960876465, "sampling/importance_sampling_ratio/min": 0.2559182047843933, "sampling/importance_sampling_ratio/mean": 1.0221158266067505, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.847815103828907, "clip_ratio/low_mean": 0.054673376493155956, "clip_ratio/low_min": 0.054673376493155956, "clip_ratio/high_mean": 0.05293074809014797, "clip_ratio/high_max": 0.05293074809014797, "clip_ratio/region_mean": 0.10760412458330393, "reward_total_mean": 0.5450050234794617, "reward_meter_mean": 0.590712308883667, "reward_meter_std": 0.4094976782798767, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9817673563957214, "reward_repeat_soft_std": 0.026286648586392403, "reward_judge_quality_mean": 0.39249998331069946, "reward_judge_quality_std": 0.13905291259288788, "reward_total_composite_mean": 0.5450050234794617, "reward_total_composite_std": 0.2797129452228546} {"timestamp_utc": "2026-04-13T02:01:30Z", "mode": "train", "global_step": 1480, "epoch": 0.1486690105474636, "loss": -0.0893, "grad_norm": 6.830164909362793, "learning_rate": 5.518181818181818e-06, "num_tokens": 2659096.0, "completions/mean_length": 91.125, "completions/min_length": 66.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.125, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.844693660736084, "rewards/meter/std": 0.32292640209198, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8994616270065308, "rewards/repeat_soft/std": 0.10280414670705795, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7460583448410034, "rewards/total_composite/std": 0.14147727191448212, "reward": 0.7460583448410034, "reward_std": 0.1414772868156433, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11813610047101974, "sampling/sampling_logp_difference/max": 1.4147779941558838, "sampling/importance_sampling_ratio/min": 0.24297954142093658, "sampling/importance_sampling_ratio/mean": 1.0255165100097656, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9982463046908379, "clip_ratio/low_mean": 0.02083333395421505, "clip_ratio/low_min": 0.02083333395421505, "clip_ratio/high_mean": 0.08710483275353909, "clip_ratio/high_max": 0.08710483275353909, "clip_ratio/region_mean": 0.10793816670775414, "reward_total_mean": 0.7460583448410034, "reward_meter_mean": 0.844693660736084, "reward_meter_std": 0.32292640209198, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8994616270065308, "reward_repeat_soft_std": 0.10280414670705795, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7460583448410034, "reward_total_composite_std": 0.14147727191448212} {"timestamp_utc": "2026-04-13T02:01:39Z", "mode": "train", "global_step": 1481, "epoch": 0.14876946258161727, "loss": 0.0319, "grad_norm": 5.510232448577881, "learning_rate": 5.515151515151515e-06, "num_tokens": 2661755.0, "completions/mean_length": 139.375, "completions/min_length": 123.0, "completions/max_length": 154.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 139.375, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 154.0, "rewards/meter/mean": 0.9909628033638, "rewards/meter/std": 0.007529743947088718, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8745059967041016, "rewards/repeat_soft/std": 0.059220653027296066, "rewards/judge_quality/mean": 0.29249998927116394, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7711338996887207, "rewards/total_composite/std": 0.029167061671614647, "reward": 0.7711338996887207, "reward_std": 0.02916708216071129, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11165149509906769, "sampling/sampling_logp_difference/max": 1.6204338073730469, "sampling/importance_sampling_ratio/min": 0.19781285524368286, "sampling/importance_sampling_ratio/mean": 1.0241178274154663, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9891822338104248, "clip_ratio/low_mean": 0.06013478431850672, "clip_ratio/low_min": 0.06013478431850672, "clip_ratio/high_mean": 0.03380992356687784, "clip_ratio/high_max": 0.03380992356687784, "clip_ratio/region_mean": 0.09394470788538456, "reward_total_mean": 0.7711338996887207, "reward_meter_mean": 0.9909628033638, "reward_meter_std": 0.007529743947088718, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8745059967041016, "reward_repeat_soft_std": 0.059220653027296066, "reward_judge_quality_mean": 0.29249998927116394, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7711338996887207, "reward_total_composite_std": 0.029167061671614647} {"timestamp_utc": "2026-04-13T02:01:47Z", "mode": "train", "global_step": 1482, "epoch": 0.14886991461577098, "loss": -0.0154, "grad_norm": 10.314318656921387, "learning_rate": 5.512121212121213e-06, "num_tokens": 2663395.0, "completions/mean_length": 42.0, "completions/min_length": 38.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.0, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.9905704259872437, "rewards/meter/std": 0.0027401656843721867, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8789701461791992, "rewards/repeat_soft/std": 0.06517865508794785, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.8032786846160889, "rewards/total_composite/std": 0.015636112540960312, "reward": 0.8032786846160889, "reward_std": 0.015636105090379715, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10933619737625122, "sampling/sampling_logp_difference/max": 1.8789544105529785, "sampling/importance_sampling_ratio/min": 0.15274974703788757, "sampling/importance_sampling_ratio/mean": 0.994135856628418, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5185326263308525, "clip_ratio/low_mean": 0.025320513173937798, "clip_ratio/low_min": 0.025320513173937798, "clip_ratio/high_mean": 0.06166206439957023, "clip_ratio/high_max": 0.06166206439957023, "clip_ratio/region_mean": 0.08698257757350802, "reward_total_mean": 0.8032786846160889, "reward_meter_mean": 0.9905704259872437, "reward_meter_std": 0.0027401656843721867, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8789701461791992, "reward_repeat_soft_std": 0.06517865508794785, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.8032786846160889, "reward_total_composite_std": 0.015636112540960312} {"timestamp_utc": "2026-04-13T02:01:54Z", "mode": "train", "global_step": 1483, "epoch": 0.14897036664992466, "loss": 0.0538, "grad_norm": 23.408658981323242, "learning_rate": 5.50909090909091e-06, "num_tokens": 2664824.0, "completions/mean_length": 25.625, "completions/min_length": 22.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.625, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9752070903778076, "rewards/meter/std": 0.03288373351097107, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.958136796951294, "rewards/repeat_soft/std": 0.009233043529093266, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8162819147109985, "rewards/total_composite/std": 0.016502495855093002, "reward": 0.8162819147109985, "reward_std": 0.0165024995803833, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12117727100849152, "sampling/sampling_logp_difference/max": 2.334135055541992, "sampling/importance_sampling_ratio/min": 0.09689425677061081, "sampling/importance_sampling_ratio/mean": 1.0148956775665283, "sampling/importance_sampling_ratio/max": 1.8424254655838013, "entropy": 0.8651778399944305, "clip_ratio/low_mean": 0.04180194903165102, "clip_ratio/low_min": 0.04180194903165102, "clip_ratio/high_mean": 0.09558688756078482, "clip_ratio/high_max": 0.09558688756078482, "clip_ratio/region_mean": 0.13738883659243584, "reward_total_mean": 0.8162819147109985, "reward_meter_mean": 0.9752070903778076, "reward_meter_std": 0.03288373351097107, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.958136796951294, "reward_repeat_soft_std": 0.009233043529093266, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8162819147109985, "reward_total_composite_std": 0.016502495855093002} {"timestamp_utc": "2026-04-13T02:02:01Z", "mode": "train", "global_step": 1484, "epoch": 0.14907081868407834, "loss": 0.0893, "grad_norm": 11.876117706298828, "learning_rate": 5.506060606060607e-06, "num_tokens": 2666511.0, "completions/mean_length": 53.875, "completions/min_length": 48.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.875, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9854632616043091, "rewards/meter/std": 0.014813334681093693, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9906753897666931, "rewards/repeat_soft/std": 0.012451897375285625, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.8770259618759155, "rewards/total_composite/std": 0.0785951241850853, "reward": 0.8770259618759155, "reward_std": 0.0785951316356659, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13785339891910553, "sampling/sampling_logp_difference/max": 1.7963110208511353, "sampling/importance_sampling_ratio/min": 0.1659097969532013, "sampling/importance_sampling_ratio/mean": 1.02916419506073, "sampling/importance_sampling_ratio/max": 1.9639981985092163, "entropy": 1.1462746635079384, "clip_ratio/low_mean": 0.07540443260222673, "clip_ratio/low_min": 0.07540443260222673, "clip_ratio/high_mean": 0.0497012585401535, "clip_ratio/high_max": 0.0497012585401535, "clip_ratio/region_mean": 0.12510569114238024, "reward_total_mean": 0.8770259618759155, "reward_meter_mean": 0.9854632616043091, "reward_meter_std": 0.014813334681093693, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9906753897666931, "reward_repeat_soft_std": 0.012451897375285625, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.8770259618759155, "reward_total_composite_std": 0.0785951241850853} {"timestamp_utc": "2026-04-13T02:02:09Z", "mode": "train", "global_step": 1485, "epoch": 0.14917127071823205, "loss": 0.0251, "grad_norm": 7.858964920043945, "learning_rate": 5.5030303030303034e-06, "num_tokens": 2668789.0, "completions/mean_length": 100.75, "completions/min_length": 94.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.75, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.9863893985748291, "rewards/meter/std": 0.007590859197080135, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8606148958206177, "rewards/repeat_soft/std": 0.12162228673696518, "rewards/judge_quality/mean": 0.2549999952316284, "rewards/judge_quality/std": 0.111867256462574, "rewards/total_composite/mean": 0.7564367651939392, "rewards/total_composite/std": 0.04294748976826668, "reward": 0.7564367651939392, "reward_std": 0.04294748231768608, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10928917676210403, "sampling/sampling_logp_difference/max": 1.4755964279174805, "sampling/importance_sampling_ratio/min": 0.2286423146724701, "sampling/importance_sampling_ratio/mean": 1.0011769533157349, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6940995790064335, "clip_ratio/low_mean": 0.023434335365891457, "clip_ratio/low_min": 0.023434335365891457, "clip_ratio/high_mean": 0.056303105782717466, "clip_ratio/high_max": 0.056303105782717466, "clip_ratio/region_mean": 0.07973744114860892, "reward_total_mean": 0.7564367651939392, "reward_meter_mean": 0.9863893985748291, "reward_meter_std": 0.007590859197080135, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8606148958206177, "reward_repeat_soft_std": 0.12162228673696518, "reward_judge_quality_mean": 0.2549999952316284, "reward_judge_quality_std": 0.111867256462574, "reward_total_composite_mean": 0.7564367651939392, "reward_total_composite_std": 0.04294748976826668} {"timestamp_utc": "2026-04-13T02:02:20Z", "mode": "train", "global_step": 1486, "epoch": 0.14927172275238573, "loss": -0.0748, "grad_norm": 5.32874059677124, "learning_rate": 5.500000000000001e-06, "num_tokens": 2670340.0, "completions/mean_length": 107.875, "completions/min_length": 42.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 50.142860412597656, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.8950088024139404, "rewards/meter/std": 0.256740540266037, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9889295101165771, "rewards/repeat_soft/std": 0.012674394994974136, "rewards/judge_quality/mean": 0.38749998807907104, "rewards/judge_quality/std": 0.20899419486522675, "rewards/total_composite/mean": 0.7585219144821167, "rewards/total_composite/std": 0.1416344940662384, "reward": 0.7585219144821167, "reward_std": 0.1416344791650772, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17305304110050201, "sampling/sampling_logp_difference/max": 1.6325082778930664, "sampling/importance_sampling_ratio/min": 0.19543874263763428, "sampling/importance_sampling_ratio/mean": 1.0260566473007202, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5507794469594955, "clip_ratio/low_mean": 0.015625, "clip_ratio/low_min": 0.015625, "clip_ratio/high_mean": 0.14775429014116526, "clip_ratio/high_max": 0.14775429014116526, "clip_ratio/region_mean": 0.16337929014116526, "reward_total_mean": 0.7585219144821167, "reward_meter_mean": 0.8950088024139404, "reward_meter_std": 0.256740540266037, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9889295101165771, "reward_repeat_soft_std": 0.012674394994974136, "reward_judge_quality_mean": 0.38749998807907104, "reward_judge_quality_std": 0.20899419486522675, "reward_total_composite_mean": 0.7585219144821167, "reward_total_composite_std": 0.1416344940662384} {"timestamp_utc": "2026-04-13T02:02:32Z", "mode": "train", "global_step": 1487, "epoch": 0.14937217478653944, "loss": -0.2332, "grad_norm": 1.745957851409912, "learning_rate": 5.496969696969697e-06, "num_tokens": 2673049.0, "completions/mean_length": 212.625, "completions/min_length": 151.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 169.85714721679688, "completions/min_terminated_length": 151.0, "completions/max_terminated_length": 193.0, "rewards/meter/mean": 0.9516177177429199, "rewards/meter/std": 0.11647137254476547, "rewards/count_adherence/mean": 0.8958333134651184, "rewards/count_adherence/std": 0.23464767634868622, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9092921018600464, "rewards/repeat_soft/std": 0.06001775711774826, "rewards/judge_quality/mean": 0.3037499785423279, "rewards/judge_quality/std": 0.14879876375198364, "rewards/total_composite/mean": 0.6704930067062378, "rewards/total_composite/std": 0.2787351608276367, "reward": 0.6704930067062378, "reward_std": 0.27873513102531433, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12498495727777481, "sampling/sampling_logp_difference/max": 1.5971755981445312, "sampling/importance_sampling_ratio/min": 0.20246756076812744, "sampling/importance_sampling_ratio/mean": 1.0016722679138184, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8158984780311584, "clip_ratio/low_mean": 0.009027778171002865, "clip_ratio/low_min": 0.009027778171002865, "clip_ratio/high_mean": 0.0843713553622365, "clip_ratio/high_max": 0.0843713553622365, "clip_ratio/region_mean": 0.09339913353323936, "reward_total_mean": 0.6704930067062378, "reward_meter_mean": 0.9516177177429199, "reward_meter_std": 0.11647137254476547, "reward_count_adherence_mean": 0.8958333134651184, "reward_count_adherence_std": 0.23464767634868622, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9092921018600464, "reward_repeat_soft_std": 0.06001775711774826, "reward_judge_quality_mean": 0.3037499785423279, "reward_judge_quality_std": 0.14879876375198364, "reward_total_composite_mean": 0.6704930067062378, "reward_total_composite_std": 0.2787351608276367} {"timestamp_utc": "2026-04-13T02:02:38Z", "mode": "train", "global_step": 1488, "epoch": 0.14947262682069312, "loss": 0.0153, "grad_norm": 23.0543212890625, "learning_rate": 5.493939393939395e-06, "num_tokens": 2674605.0, "completions/mean_length": 35.5, "completions/min_length": 33.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.5, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.9077030420303345, "rewards/meter/std": 0.15469977259635925, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9859945774078369, "rewards/repeat_soft/std": 0.016768986359238625, "rewards/judge_quality/mean": 0.47999998927116394, "rewards/judge_quality/std": 0.14471648633480072, "rewards/total_composite/mean": 0.8010658025741577, "rewards/total_composite/std": 0.08476328104734421, "reward": 0.8010658025741577, "reward_std": 0.08476326614618301, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1185491532087326, "sampling/sampling_logp_difference/max": 1.3404741287231445, "sampling/importance_sampling_ratio/min": 0.26172155141830444, "sampling/importance_sampling_ratio/mean": 1.0206706523895264, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7808212637901306, "clip_ratio/low_mean": 0.03566919267177582, "clip_ratio/low_min": 0.03566919267177582, "clip_ratio/high_mean": 0.08951705601066351, "clip_ratio/high_max": 0.08951705601066351, "clip_ratio/region_mean": 0.12518624868243933, "reward_total_mean": 0.8010658025741577, "reward_meter_mean": 0.9077030420303345, "reward_meter_std": 0.15469977259635925, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9859945774078369, "reward_repeat_soft_std": 0.016768986359238625, "reward_judge_quality_mean": 0.47999998927116394, "reward_judge_quality_std": 0.14471648633480072, "reward_total_composite_mean": 0.8010658025741577, "reward_total_composite_std": 0.08476328104734421} {"timestamp_utc": "2026-04-13T02:02:45Z", "mode": "train", "global_step": 1489, "epoch": 0.1495730788548468, "loss": 0.0607, "grad_norm": 10.985793113708496, "learning_rate": 5.490909090909091e-06, "num_tokens": 2676428.0, "completions/mean_length": 62.875, "completions/min_length": 52.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.875, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9921385049819946, "rewards/meter/std": 0.0014869621954858303, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8647097945213318, "rewards/repeat_soft/std": 0.0434577502310276, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.8025583028793335, "rewards/total_composite/std": 0.01849461905658245, "reward": 0.8025583028793335, "reward_std": 0.018494615331292152, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09936192631721497, "sampling/sampling_logp_difference/max": 1.5356321334838867, "sampling/importance_sampling_ratio/min": 0.21531954407691956, "sampling/importance_sampling_ratio/mean": 1.0111145973205566, "sampling/importance_sampling_ratio/max": 1.8867636919021606, "entropy": 0.6459084488451481, "clip_ratio/low_mean": 0.016304347664117813, "clip_ratio/low_min": 0.016304347664117813, "clip_ratio/high_mean": 0.09266381920315325, "clip_ratio/high_max": 0.09266381920315325, "clip_ratio/region_mean": 0.10896816686727107, "reward_total_mean": 0.8025583028793335, "reward_meter_mean": 0.9921385049819946, "reward_meter_std": 0.0014869621954858303, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8647097945213318, "reward_repeat_soft_std": 0.0434577502310276, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.8025583028793335, "reward_total_composite_std": 0.01849461905658245} {"timestamp_utc": "2026-04-13T02:02:51Z", "mode": "train", "global_step": 1490, "epoch": 0.1496735308890005, "loss": 0.0725, "grad_norm": 22.2070369720459, "learning_rate": 5.487878787878789e-06, "num_tokens": 2677858.0, "completions/mean_length": 24.75, "completions/min_length": 21.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.75, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9237802624702454, "rewards/meter/std": 0.15832525491714478, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.956274688243866, "rewards/repeat_soft/std": 0.012105435132980347, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.7918286323547363, "rewards/total_composite/std": 0.06954976916313171, "reward": 0.7918286323547363, "reward_std": 0.06954976171255112, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12166071683168411, "sampling/sampling_logp_difference/max": 1.1689605712890625, "sampling/importance_sampling_ratio/min": 0.31068968772888184, "sampling/importance_sampling_ratio/mean": 1.046405553817749, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9962312206625938, "clip_ratio/low_mean": 0.0223214291036129, "clip_ratio/low_min": 0.0223214291036129, "clip_ratio/high_mean": 0.08107796031981707, "clip_ratio/high_max": 0.08107796031981707, "clip_ratio/region_mean": 0.10339938942342997, "reward_total_mean": 0.7918286323547363, "reward_meter_mean": 0.9237802624702454, "reward_meter_std": 0.15832525491714478, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.956274688243866, "reward_repeat_soft_std": 0.012105435132980347, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.7918286323547363, "reward_total_composite_std": 0.06954976916313171} {"timestamp_utc": "2026-04-13T02:03:03Z", "mode": "train", "global_step": 1491, "epoch": 0.1497739829231542, "loss": -0.0875, "grad_norm": 1.7355382442474365, "learning_rate": 5.484848484848485e-06, "num_tokens": 2679367.0, "completions/mean_length": 84.625, "completions/min_length": 18.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 23.571430206298828, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.8307275772094727, "rewards/meter/std": 0.3308800458908081, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.13617216050624847, "rewards/total_composite/mean": 0.7012832164764404, "rewards/total_composite/std": 0.2848622798919678, "reward": 0.7012832164764404, "reward_std": 0.2848622798919678, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1647750735282898, "sampling/sampling_logp_difference/max": 1.643721580505371, "sampling/importance_sampling_ratio/min": 0.19325946271419525, "sampling/importance_sampling_ratio/mean": 1.0513558387756348, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0949913561344147, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.1404590029269457, "clip_ratio/high_max": 0.1404590029269457, "clip_ratio/region_mean": 0.1404590029269457, "reward_total_mean": 0.7012832164764404, "reward_meter_mean": 0.8307275772094727, "reward_meter_std": 0.3308800458908081, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.13617216050624847, "reward_total_composite_mean": 0.7012832164764404, "reward_total_composite_std": 0.2848622798919678} {"timestamp_utc": "2026-04-13T02:03:09Z", "mode": "train", "global_step": 1492, "epoch": 0.14987443495730787, "loss": -0.0133, "grad_norm": 9.943988800048828, "learning_rate": 5.4818181818181825e-06, "num_tokens": 2681208.0, "completions/mean_length": 58.125, "completions/min_length": 51.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.125, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.989223837852478, "rewards/meter/std": 0.008124805055558681, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9692332744598389, "rewards/repeat_soft/std": 0.05370841175317764, "rewards/judge_quality/mean": 0.42124998569488525, "rewards/judge_quality/std": 0.06998724490404129, "rewards/total_composite/mean": 0.818449079990387, "rewards/total_composite/std": 0.0206146277487278, "reward": 0.818449079990387, "reward_std": 0.020614637061953545, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12814363837242126, "sampling/sampling_logp_difference/max": 1.35711669921875, "sampling/importance_sampling_ratio/min": 0.36230209469795227, "sampling/importance_sampling_ratio/mean": 0.9985095858573914, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0042292773723602, "clip_ratio/low_mean": 0.030614973045885563, "clip_ratio/low_min": 0.030614973045885563, "clip_ratio/high_mean": 0.11643979977816343, "clip_ratio/high_max": 0.11643979977816343, "clip_ratio/region_mean": 0.147054772824049, "reward_total_mean": 0.818449079990387, "reward_meter_mean": 0.989223837852478, "reward_meter_std": 0.008124805055558681, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9692332744598389, "reward_repeat_soft_std": 0.05370841175317764, "reward_judge_quality_mean": 0.42124998569488525, "reward_judge_quality_std": 0.06998724490404129, "reward_total_composite_mean": 0.818449079990387, "reward_total_composite_std": 0.0206146277487278} {"timestamp_utc": "2026-04-13T02:03:21Z", "mode": "train", "global_step": 1493, "epoch": 0.14997488699146158, "loss": -0.168, "grad_norm": 2.522711992263794, "learning_rate": 5.478787878787879e-06, "num_tokens": 2683102.0, "completions/mean_length": 132.75, "completions/min_length": 68.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 78.5714340209961, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.9079997539520264, "rewards/meter/std": 0.22432391345500946, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9547609090805054, "rewards/repeat_soft/std": 0.02108731120824814, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.1345893144607544, "rewards/total_composite/mean": 0.6602760553359985, "rewards/total_composite/std": 0.2908209264278412, "reward": 0.6602760553359985, "reward_std": 0.2908209562301636, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1388915479183197, "sampling/sampling_logp_difference/max": 1.4445304870605469, "sampling/importance_sampling_ratio/min": 0.2358568012714386, "sampling/importance_sampling_ratio/mean": 1.0052987337112427, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8705780729651451, "clip_ratio/low_mean": 0.009868420660495758, "clip_ratio/low_min": 0.009868420660495758, "clip_ratio/high_mean": 0.09319112729281187, "clip_ratio/high_max": 0.09319112729281187, "clip_ratio/region_mean": 0.10305954795330763, "reward_total_mean": 0.6602760553359985, "reward_meter_mean": 0.9079997539520264, "reward_meter_std": 0.22432391345500946, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9547609090805054, "reward_repeat_soft_std": 0.02108731120824814, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.1345893144607544, "reward_total_composite_mean": 0.6602760553359985, "reward_total_composite_std": 0.2908209264278412} {"timestamp_utc": "2026-04-13T02:03:33Z", "mode": "train", "global_step": 1494, "epoch": 0.15007533902561526, "loss": -0.1068, "grad_norm": 2.7360422611236572, "learning_rate": 5.475757575757576e-06, "num_tokens": 2684674.0, "completions/mean_length": 107.5, "completions/min_length": 42.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 49.71428680419922, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.7507380247116089, "rewards/meter/std": 0.35729268193244934, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9899272322654724, "rewards/repeat_soft/std": 0.008078210055828094, "rewards/judge_quality/mean": 0.4987500309944153, "rewards/judge_quality/std": 0.28965190052986145, "rewards/total_composite/mean": 0.6664319038391113, "rewards/total_composite/std": 0.3330219089984894, "reward": 0.6664319038391113, "reward_std": 0.333021879196167, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11489535868167877, "sampling/sampling_logp_difference/max": 1.4275727272033691, "sampling/importance_sampling_ratio/min": 0.23989050090312958, "sampling/importance_sampling_ratio/mean": 1.0256102085113525, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47871339693665504, "clip_ratio/low_mean": 0.03449292480945587, "clip_ratio/low_min": 0.03449292480945587, "clip_ratio/high_mean": 0.06701393006369472, "clip_ratio/high_max": 0.06701393006369472, "clip_ratio/region_mean": 0.10150685487315059, "reward_total_mean": 0.6664319038391113, "reward_meter_mean": 0.7507380247116089, "reward_meter_std": 0.35729268193244934, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9899272322654724, "reward_repeat_soft_std": 0.008078210055828094, "reward_judge_quality_mean": 0.4987500309944153, "reward_judge_quality_std": 0.28965190052986145, "reward_total_composite_mean": 0.6664319038391113, "reward_total_composite_std": 0.3330219089984894} {"timestamp_utc": "2026-04-13T02:03:40Z", "mode": "train", "global_step": 1495, "epoch": 0.15017579105976897, "loss": 0.0405, "grad_norm": 12.788679122924805, "learning_rate": 5.472727272727273e-06, "num_tokens": 2686286.0, "completions/mean_length": 49.5, "completions/min_length": 44.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.5, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9177952408790588, "rewards/meter/std": 0.1854632943868637, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9418348670005798, "rewards/repeat_soft/std": 0.05153948441147804, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7888163328170776, "rewards/total_composite/std": 0.08427172899246216, "reward": 0.7888163328170776, "reward_std": 0.08427172154188156, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12314476817846298, "sampling/sampling_logp_difference/max": 1.6899523735046387, "sampling/importance_sampling_ratio/min": 0.18452832102775574, "sampling/importance_sampling_ratio/mean": 1.0262824296951294, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6970284879207611, "clip_ratio/low_mean": 0.01179245300590992, "clip_ratio/low_min": 0.01179245300590992, "clip_ratio/high_mean": 0.08729096921160817, "clip_ratio/high_max": 0.08729096921160817, "clip_ratio/region_mean": 0.09908342221751809, "reward_total_mean": 0.7888163328170776, "reward_meter_mean": 0.9177952408790588, "reward_meter_std": 0.1854632943868637, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9418348670005798, "reward_repeat_soft_std": 0.05153948441147804, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7888163328170776, "reward_total_composite_std": 0.08427172899246216} {"timestamp_utc": "2026-04-13T02:03:51Z", "mode": "train", "global_step": 1496, "epoch": 0.15027624309392265, "loss": -0.1406, "grad_norm": 1.9185426235198975, "learning_rate": 5.469696969696971e-06, "num_tokens": 2688010.0, "completions/mean_length": 171.5, "completions/min_length": 52.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.6366961002349854, "rewards/meter/std": 0.4263046085834503, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9674922227859497, "rewards/repeat_soft/std": 0.041211992502212524, "rewards/judge_quality/mean": 0.26875001192092896, "rewards/judge_quality/std": 0.17715509235858917, "rewards/total_composite/mean": 0.5498279333114624, "rewards/total_composite/std": 0.34667372703552246, "reward": 0.5498279333114624, "reward_std": 0.3466736972332001, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17895382642745972, "sampling/sampling_logp_difference/max": 1.1177401542663574, "sampling/importance_sampling_ratio/min": 0.32701796293258667, "sampling/importance_sampling_ratio/mean": 1.0405454635620117, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6919710785150528, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10843618307262659, "clip_ratio/high_max": 0.10843618307262659, "clip_ratio/region_mean": 0.10843618307262659, "reward_total_mean": 0.5498279333114624, "reward_meter_mean": 0.6366961002349854, "reward_meter_std": 0.4263046085834503, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9674922227859497, "reward_repeat_soft_std": 0.041211992502212524, "reward_judge_quality_mean": 0.26875001192092896, "reward_judge_quality_std": 0.17715509235858917, "reward_total_composite_mean": 0.5498279333114624, "reward_total_composite_std": 0.34667372703552246} {"timestamp_utc": "2026-04-13T02:03:57Z", "mode": "train", "global_step": 1497, "epoch": 0.15037669512807633, "loss": 0.0566, "grad_norm": 20.655818939208984, "learning_rate": 5.466666666666667e-06, "num_tokens": 2689413.0, "completions/mean_length": 26.375, "completions/min_length": 24.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.375, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.9899139404296875, "rewards/meter/std": 0.007317697163671255, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.38875001668930054, "rewards/judge_quality/std": 0.08675704896450043, "rewards/total_composite/mean": 0.8083362579345703, "rewards/total_composite/std": 0.02526802569627762, "reward": 0.8083362579345703, "reward_std": 0.025268014520406723, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11687242239713669, "sampling/sampling_logp_difference/max": 1.4989581108093262, "sampling/importance_sampling_ratio/min": 0.2233627587556839, "sampling/importance_sampling_ratio/mean": 1.0313706398010254, "sampling/importance_sampling_ratio/max": 1.848522663116455, "entropy": 0.8368498459458351, "clip_ratio/low_mean": 0.04230769397690892, "clip_ratio/low_min": 0.04230769397690892, "clip_ratio/high_mean": 0.06739862682297826, "clip_ratio/high_max": 0.06739862682297826, "clip_ratio/region_mean": 0.10970632079988718, "reward_total_mean": 0.8083362579345703, "reward_meter_mean": 0.9899139404296875, "reward_meter_std": 0.007317697163671255, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.38875001668930054, "reward_judge_quality_std": 0.08675704896450043, "reward_total_composite_mean": 0.8083362579345703, "reward_total_composite_std": 0.02526802569627762} {"timestamp_utc": "2026-04-13T02:04:04Z", "mode": "train", "global_step": 1498, "epoch": 0.15047714716223004, "loss": 0.012, "grad_norm": 9.3164644241333, "learning_rate": 5.463636363636364e-06, "num_tokens": 2690946.0, "completions/mean_length": 49.625, "completions/min_length": 44.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.625, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.8969252705574036, "rewards/meter/std": 0.21242868900299072, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.936497151851654, "rewards/repeat_soft/std": 0.03142538294196129, "rewards/judge_quality/mean": 0.5862500071525574, "rewards/judge_quality/std": 0.22984080016613007, "rewards/total_composite/mean": 0.8231410980224609, "rewards/total_composite/std": 0.12828466296195984, "reward": 0.8231410980224609, "reward_std": 0.12828467786312103, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11736145615577698, "sampling/sampling_logp_difference/max": 1.5762815475463867, "sampling/importance_sampling_ratio/min": 0.20674243569374084, "sampling/importance_sampling_ratio/mean": 1.0267114639282227, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.759080808609724, "clip_ratio/low_mean": 0.052071737591177225, "clip_ratio/low_min": 0.052071737591177225, "clip_ratio/high_mean": 0.03907504538074136, "clip_ratio/high_max": 0.03907504538074136, "clip_ratio/region_mean": 0.09114678297191858, "reward_total_mean": 0.8231410980224609, "reward_meter_mean": 0.8969252705574036, "reward_meter_std": 0.21242868900299072, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.936497151851654, "reward_repeat_soft_std": 0.03142538294196129, "reward_judge_quality_mean": 0.5862500071525574, "reward_judge_quality_std": 0.22984080016613007, "reward_total_composite_mean": 0.8231410980224609, "reward_total_composite_std": 0.12828466296195984} {"timestamp_utc": "2026-04-13T02:04:11Z", "mode": "train", "global_step": 1499, "epoch": 0.15057759919638372, "loss": 0.0074, "grad_norm": 10.757710456848145, "learning_rate": 5.460606060606061e-06, "num_tokens": 2692840.0, "completions/mean_length": 73.75, "completions/min_length": 68.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.75, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.827200710773468, "rewards/meter/std": 0.2531125247478485, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9418743848800659, "rewards/repeat_soft/std": 0.03400443121790886, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7296777963638306, "rewards/total_composite/std": 0.12609045207500458, "reward": 0.7296777963638306, "reward_std": 0.12609043717384338, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12001737207174301, "sampling/sampling_logp_difference/max": 1.7451720237731934, "sampling/importance_sampling_ratio/min": 0.17461493611335754, "sampling/importance_sampling_ratio/mean": 0.9997701048851013, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7471847161650658, "clip_ratio/low_mean": 0.029960318468511105, "clip_ratio/low_min": 0.029960318468511105, "clip_ratio/high_mean": 0.08414205117151141, "clip_ratio/high_max": 0.08414205117151141, "clip_ratio/region_mean": 0.11410236964002252, "reward_total_mean": 0.7296777963638306, "reward_meter_mean": 0.827200710773468, "reward_meter_std": 0.2531125247478485, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9418743848800659, "reward_repeat_soft_std": 0.03400443121790886, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7296777963638306, "reward_total_composite_std": 0.12609045207500458} {"timestamp_utc": "2026-04-13T02:04:18Z", "mode": "train", "global_step": 1500, "epoch": 0.15067805123053743, "loss": -0.0052, "grad_norm": 11.868607521057129, "learning_rate": 5.457575757575758e-06, "num_tokens": 2694459.0, "completions/mean_length": 50.375, "completions/min_length": 43.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.375, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9241307973861694, "rewards/meter/std": 0.15917173027992249, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9773482084274292, "rewards/repeat_soft/std": 0.03203749656677246, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.8668437004089355, "rewards/total_composite/std": 0.1250350922346115, "reward": 0.8668437004089355, "reward_std": 0.12503507733345032, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13263677060604095, "sampling/sampling_logp_difference/max": 2.9111547470092773, "sampling/importance_sampling_ratio/min": 0.05441286414861679, "sampling/importance_sampling_ratio/mean": 1.0215086936950684, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9419487938284874, "clip_ratio/low_mean": 0.06739429105073214, "clip_ratio/low_min": 0.06739429105073214, "clip_ratio/high_mean": 0.0646477323025465, "clip_ratio/high_max": 0.0646477323025465, "clip_ratio/region_mean": 0.13204202335327864, "reward_total_mean": 0.8668437004089355, "reward_meter_mean": 0.9241307973861694, "reward_meter_std": 0.15917173027992249, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9773482084274292, "reward_repeat_soft_std": 0.03203749656677246, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.8668437004089355, "reward_total_composite_std": 0.1250350922346115} {"timestamp_utc": "2026-04-13T02:05:02Z", "mode": "eval", "global_step": 1500, "epoch": 0.15067805123053743, "eval_loss": NaN, "eval_runtime": 43.074, "eval_samples_per_second": 1.857, "eval_steps_per_second": 0.232, "eval_num_tokens": 2694459.0, "eval_completions/mean_length": 79.525, "eval_completions/min_length": 35.6, "eval_completions/max_length": 130.7, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 79.525, "eval_completions/min_terminated_length": 35.6, "eval_completions/max_terminated_length": 130.7, "eval_rewards/meter/mean": 0.9272477805614472, "eval_rewards/meter/std": 0.12010596450418234, "eval_rewards/count_adherence/mean": 0.9979166626930237, "eval_rewards/count_adherence/std": 0.00589255727827549, "eval_rewards/hard_gate/mean": 1.0, "eval_rewards/hard_gate/std": 0.0, "eval_rewards/repeat_soft/mean": 0.8996355831623077, "eval_rewards/repeat_soft/std": 0.08510150499641896, "eval_rewards/judge_quality/mean": 0.409499990940094, "eval_rewards/judge_quality/std": 0.13192110881209373, "eval_rewards/total_composite/mean": 0.7797625660896301, "eval_rewards/total_composite/std": 0.0698787909001112, "eval_reward": 0.7797625660896301, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.06328675486147403, "eval_sampling/sampling_logp_difference/max": 1.0764353275299072, "eval_sampling/importance_sampling_ratio/min": 0.35324908792972565, "eval_sampling/importance_sampling_ratio/mean": 1.0157706260681152, "eval_sampling/importance_sampling_ratio/max": 1.4055296778678894, "eval_entropy": 0.7148518800735474, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7797625660896301, "eval_reward_meter_mean": 0.9272477805614472, "eval_reward_meter_std": 0.12010596450418234, "eval_reward_count_adherence_mean": 0.9979166626930237, "eval_reward_count_adherence_std": 0.00589255727827549, "eval_reward_hard_gate_mean": 1.0, "eval_reward_hard_gate_std": 0.0, "eval_reward_repeat_soft_mean": 0.8996355831623077, "eval_reward_repeat_soft_std": 0.08510150499641896, "eval_reward_judge_quality_mean": 0.409499990940094, "eval_reward_judge_quality_std": 0.13192110881209373, "eval_reward_total_composite_mean": 0.7797625660896301, "eval_reward_total_composite_std": 0.0698787909001112} {"timestamp_utc": "2026-04-13T02:05:15Z", "mode": "train", "global_step": 1501, "epoch": 0.1507785032646911, "loss": -0.0072, "grad_norm": 7.503206729888916, "learning_rate": 5.4545454545454545e-06, "num_tokens": 2697123.0, "completions/mean_length": 132.0, "completions/min_length": 119.0, "completions/max_length": 157.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.0, "completions/min_terminated_length": 119.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.9908676147460938, "rewards/meter/std": 0.005628687329590321, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8989278674125671, "rewards/repeat_soft/std": 0.041991714388132095, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7889082431793213, "rewards/total_composite/std": 0.034391582012176514, "reward": 0.7889082431793213, "reward_std": 0.03439158946275711, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12746398150920868, "sampling/sampling_logp_difference/max": 2.4017248153686523, "sampling/importance_sampling_ratio/min": 0.09056161344051361, "sampling/importance_sampling_ratio/mean": 1.0170341730117798, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9469921141862869, "clip_ratio/low_mean": 0.033661289140582085, "clip_ratio/low_min": 0.033661289140582085, "clip_ratio/high_mean": 0.0787984412163496, "clip_ratio/high_max": 0.0787984412163496, "clip_ratio/region_mean": 0.11245973035693169, "reward_total_mean": 0.7889082431793213, "reward_meter_mean": 0.9908676147460938, "reward_meter_std": 0.005628687329590321, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8989278674125671, "reward_repeat_soft_std": 0.041991714388132095, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7889082431793213, "reward_total_composite_std": 0.034391582012176514} {"timestamp_utc": "2026-04-13T02:05:21Z", "mode": "train", "global_step": 1502, "epoch": 0.1508789552988448, "loss": 0.0458, "grad_norm": 10.7941255569458, "learning_rate": 5.451515151515152e-06, "num_tokens": 2698743.0, "completions/mean_length": 44.5, "completions/min_length": 41.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.5, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.9743812084197998, "rewards/meter/std": 0.021657386794686317, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.93230140209198, "rewards/repeat_soft/std": 0.025439850986003876, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.8122016787528992, "rewards/total_composite/std": 0.010486184619367123, "reward": 0.8122016787528992, "reward_std": 0.010486162267625332, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1054442971944809, "sampling/sampling_logp_difference/max": 1.1654598712921143, "sampling/importance_sampling_ratio/min": 0.3117792308330536, "sampling/importance_sampling_ratio/mean": 1.034696340560913, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8077494278550148, "clip_ratio/low_mean": 0.038491460494697094, "clip_ratio/low_min": 0.038491460494697094, "clip_ratio/high_mean": 0.04600446345284581, "clip_ratio/high_max": 0.04600446345284581, "clip_ratio/region_mean": 0.0844959239475429, "reward_total_mean": 0.8122016787528992, "reward_meter_mean": 0.9743812084197998, "reward_meter_std": 0.021657386794686317, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.93230140209198, "reward_repeat_soft_std": 0.025439850986003876, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.8122016787528992, "reward_total_composite_std": 0.010486184619367123} {"timestamp_utc": "2026-04-13T02:05:28Z", "mode": "train", "global_step": 1503, "epoch": 0.1509794073329985, "loss": -0.0029, "grad_norm": 13.466681480407715, "learning_rate": 5.448484848484848e-06, "num_tokens": 2700625.0, "completions/mean_length": 55.25, "completions/min_length": 50.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.25, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9491061568260193, "rewards/meter/std": 0.05505192279815674, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9639427661895752, "rewards/repeat_soft/std": 0.039065029472112656, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8051170110702515, "rewards/total_composite/std": 0.027308857068419456, "reward": 0.8051170110702515, "reward_std": 0.027308857068419456, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10971613228321075, "sampling/sampling_logp_difference/max": 1.0427894592285156, "sampling/importance_sampling_ratio/min": 0.35247012972831726, "sampling/importance_sampling_ratio/mean": 1.0053340196609497, "sampling/importance_sampling_ratio/max": 1.687536597251892, "entropy": 0.8560227528214455, "clip_ratio/low_mean": 0.03302813787013292, "clip_ratio/low_min": 0.03302813787013292, "clip_ratio/high_mean": 0.06523342244327068, "clip_ratio/high_max": 0.06523342244327068, "clip_ratio/region_mean": 0.0982615603134036, "reward_total_mean": 0.8051170110702515, "reward_meter_mean": 0.9491061568260193, "reward_meter_std": 0.05505192279815674, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9639427661895752, "reward_repeat_soft_std": 0.039065029472112656, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8051170110702515, "reward_total_composite_std": 0.027308857068419456} {"timestamp_utc": "2026-04-13T02:05:36Z", "mode": "train", "global_step": 1504, "epoch": 0.15107985936715218, "loss": 0.0253, "grad_norm": 8.589951515197754, "learning_rate": 5.445454545454546e-06, "num_tokens": 2702848.0, "completions/mean_length": 106.875, "completions/min_length": 101.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.875, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.9887218475341797, "rewards/meter/std": 0.014599796384572983, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8911319971084595, "rewards/repeat_soft/std": 0.079526387155056, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.8036630749702454, "rewards/total_composite/std": 0.022318996489048004, "reward": 0.8036630749702454, "reward_std": 0.022318996489048004, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1117067039012909, "sampling/sampling_logp_difference/max": 1.7715363502502441, "sampling/importance_sampling_ratio/min": 0.17007149755954742, "sampling/importance_sampling_ratio/mean": 1.007131576538086, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7908472418785095, "clip_ratio/low_mean": 0.05514128180220723, "clip_ratio/low_min": 0.05514128180220723, "clip_ratio/high_mean": 0.061379000544548035, "clip_ratio/high_max": 0.061379000544548035, "clip_ratio/region_mean": 0.11652028234675527, "reward_total_mean": 0.8036630749702454, "reward_meter_mean": 0.9887218475341797, "reward_meter_std": 0.014599796384572983, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8911319971084595, "reward_repeat_soft_std": 0.079526387155056, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.8036630749702454, "reward_total_composite_std": 0.022318996489048004} {"timestamp_utc": "2026-04-13T02:05:42Z", "mode": "train", "global_step": 1505, "epoch": 0.1511803114013059, "loss": 0.0251, "grad_norm": 17.672468185424805, "learning_rate": 5.442424242424243e-06, "num_tokens": 2704402.0, "completions/mean_length": 33.25, "completions/min_length": 29.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.25, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.9415143728256226, "rewards/meter/std": 0.02594984509050846, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9509804248809814, "rewards/repeat_soft/std": 0.036470428109169006, "rewards/judge_quality/mean": 0.3675000071525574, "rewards/judge_quality/std": 0.09808888286352158, "rewards/total_composite/mean": 0.7790294885635376, "rewards/total_composite/std": 0.024891948327422142, "reward": 0.7790294885635376, "reward_std": 0.024891942739486694, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1118684783577919, "sampling/sampling_logp_difference/max": 1.2430381774902344, "sampling/importance_sampling_ratio/min": 0.2885063588619232, "sampling/importance_sampling_ratio/mean": 1.0033100843429565, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5974725931882858, "clip_ratio/low_mean": 0.03676470695063472, "clip_ratio/low_min": 0.03676470695063472, "clip_ratio/high_mean": 0.05624496238306165, "clip_ratio/high_max": 0.05624496238306165, "clip_ratio/region_mean": 0.09300966933369637, "reward_total_mean": 0.7790294885635376, "reward_meter_mean": 0.9415143728256226, "reward_meter_std": 0.02594984509050846, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9509804248809814, "reward_repeat_soft_std": 0.036470428109169006, "reward_judge_quality_mean": 0.3675000071525574, "reward_judge_quality_std": 0.09808888286352158, "reward_total_composite_mean": 0.7790294885635376, "reward_total_composite_std": 0.024891948327422142} {"timestamp_utc": "2026-04-13T02:05:50Z", "mode": "train", "global_step": 1506, "epoch": 0.15128076343545957, "loss": -0.0091, "grad_norm": 7.422491550445557, "learning_rate": 5.43939393939394e-06, "num_tokens": 2706766.0, "completions/mean_length": 111.5, "completions/min_length": 100.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.5, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.9765684604644775, "rewards/meter/std": 0.023793641477823257, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9137105941772461, "rewards/repeat_soft/std": 0.057420194149017334, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.8004518747329712, "rewards/total_composite/std": 0.023968910798430443, "reward": 0.8004518747329712, "reward_std": 0.02396891824901104, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11790117621421814, "sampling/sampling_logp_difference/max": 1.7297474145889282, "sampling/importance_sampling_ratio/min": 0.1773291975259781, "sampling/importance_sampling_ratio/mean": 1.0094729661941528, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8127340450882912, "clip_ratio/low_mean": 0.037101841531693935, "clip_ratio/low_min": 0.037101841531693935, "clip_ratio/high_mean": 0.0837528333067894, "clip_ratio/high_max": 0.0837528333067894, "clip_ratio/region_mean": 0.12085467483848333, "reward_total_mean": 0.8004518747329712, "reward_meter_mean": 0.9765684604644775, "reward_meter_std": 0.023793641477823257, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9137105941772461, "reward_repeat_soft_std": 0.057420194149017334, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.8004518747329712, "reward_total_composite_std": 0.023968910798430443} {"timestamp_utc": "2026-04-13T02:05:56Z", "mode": "train", "global_step": 1507, "epoch": 0.15138121546961325, "loss": -0.0407, "grad_norm": 7.667252540588379, "learning_rate": 5.436363636363636e-06, "num_tokens": 2708896.0, "completions/mean_length": 88.25, "completions/min_length": 73.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 88.25, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.9867131114006042, "rewards/meter/std": 0.009960981085896492, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8135758638381958, "rewards/repeat_soft/std": 0.13864843547344208, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7886284589767456, "rewards/total_composite/std": 0.03287920355796814, "reward": 0.7886284589767456, "reward_std": 0.032879214733839035, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09784458577632904, "sampling/sampling_logp_difference/max": 1.2445645332336426, "sampling/importance_sampling_ratio/min": 0.2880663275718689, "sampling/importance_sampling_ratio/mean": 1.0097434520721436, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.615460742264986, "clip_ratio/low_mean": 0.030827971640974283, "clip_ratio/low_min": 0.030827971640974283, "clip_ratio/high_mean": 0.0802142284810543, "clip_ratio/high_max": 0.0802142284810543, "clip_ratio/region_mean": 0.11104220012202859, "reward_total_mean": 0.7886284589767456, "reward_meter_mean": 0.9867131114006042, "reward_meter_std": 0.009960981085896492, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8135758638381958, "reward_repeat_soft_std": 0.13864843547344208, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7886284589767456, "reward_total_composite_std": 0.03287920355796814} {"timestamp_utc": "2026-04-13T02:06:03Z", "mode": "train", "global_step": 1508, "epoch": 0.15148166750376696, "loss": 0.0536, "grad_norm": 13.135133743286133, "learning_rate": 5.4333333333333335e-06, "num_tokens": 2710633.0, "completions/mean_length": 48.125, "completions/min_length": 42.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.125, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.9190822243690491, "rewards/meter/std": 0.13211016356945038, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9590613842010498, "rewards/repeat_soft/std": 0.03211641311645508, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.0843462198972702, "rewards/total_composite/mean": 0.7749931812286377, "rewards/total_composite/std": 0.07141357660293579, "reward": 0.7749931812286377, "reward_std": 0.07141358405351639, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11760498583316803, "sampling/sampling_logp_difference/max": 1.458641529083252, "sampling/importance_sampling_ratio/min": 0.23255197703838348, "sampling/importance_sampling_ratio/mean": 1.0286953449249268, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9305659309029579, "clip_ratio/low_mean": 0.03507965710014105, "clip_ratio/low_min": 0.03507965710014105, "clip_ratio/high_mean": 0.07757146889343858, "clip_ratio/high_max": 0.07757146889343858, "clip_ratio/region_mean": 0.11265112599357963, "reward_total_mean": 0.7749931812286377, "reward_meter_mean": 0.9190822243690491, "reward_meter_std": 0.13211016356945038, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9590613842010498, "reward_repeat_soft_std": 0.03211641311645508, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.0843462198972702, "reward_total_composite_mean": 0.7749931812286377, "reward_total_composite_std": 0.07141357660293579} {"timestamp_utc": "2026-04-13T02:06:10Z", "mode": "train", "global_step": 1509, "epoch": 0.15158211953792064, "loss": 0.0114, "grad_norm": 7.274053573608398, "learning_rate": 5.430303030303032e-06, "num_tokens": 2712965.0, "completions/mean_length": 100.5, "completions/min_length": 92.0, "completions/max_length": 113.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.5, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.7882372736930847, "rewards/meter/std": 0.2907833755016327, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9117079377174377, "rewards/repeat_soft/std": 0.0599222369492054, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.7706276178359985, "rewards/total_composite/std": 0.12325190007686615, "reward": 0.7706276178359985, "reward_std": 0.12325190007686615, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10105269402265549, "sampling/sampling_logp_difference/max": 1.7320752143859863, "sampling/importance_sampling_ratio/min": 0.17691688239574432, "sampling/importance_sampling_ratio/mean": 1.0133568048477173, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5635080337524414, "clip_ratio/low_mean": 0.016151961404830217, "clip_ratio/low_min": 0.016151961404830217, "clip_ratio/high_mean": 0.0775968637317419, "clip_ratio/high_max": 0.0775968637317419, "clip_ratio/region_mean": 0.09374882513657212, "reward_total_mean": 0.7706276178359985, "reward_meter_mean": 0.7882372736930847, "reward_meter_std": 0.2907833755016327, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9117079377174377, "reward_repeat_soft_std": 0.0599222369492054, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.7706276178359985, "reward_total_composite_std": 0.12325190007686615} {"timestamp_utc": "2026-04-13T02:06:16Z", "mode": "train", "global_step": 1510, "epoch": 0.15168257157207435, "loss": -0.0002, "grad_norm": 15.804697036743164, "learning_rate": 5.427272727272728e-06, "num_tokens": 2714413.0, "completions/mean_length": 27.0, "completions/min_length": 25.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.0, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.8288131952285767, "rewards/meter/std": 0.3325580060482025, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9479215145111084, "rewards/repeat_soft/std": 0.023002946749329567, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7471331357955933, "rewards/total_composite/std": 0.15268079936504364, "reward": 0.7471331357955933, "reward_std": 0.15268079936504364, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13922806084156036, "sampling/sampling_logp_difference/max": 1.730318546295166, "sampling/importance_sampling_ratio/min": 0.1772279441356659, "sampling/importance_sampling_ratio/mean": 1.0159852504730225, "sampling/importance_sampling_ratio/max": 1.8535408973693848, "entropy": 0.7590174823999405, "clip_ratio/low_mean": 0.024999999441206455, "clip_ratio/low_min": 0.024999999441206455, "clip_ratio/high_mean": 0.08147302456200123, "clip_ratio/high_max": 0.08147302456200123, "clip_ratio/region_mean": 0.10647302400320768, "reward_total_mean": 0.7471331357955933, "reward_meter_mean": 0.8288131952285767, "reward_meter_std": 0.3325580060482025, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9479215145111084, "reward_repeat_soft_std": 0.023002946749329567, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7471331357955933, "reward_total_composite_std": 0.15268079936504364} {"timestamp_utc": "2026-04-13T02:06:22Z", "mode": "train", "global_step": 1511, "epoch": 0.15178302360622803, "loss": 0.0269, "grad_norm": 14.362140655517578, "learning_rate": 5.424242424242425e-06, "num_tokens": 2715853.0, "completions/mean_length": 45.0, "completions/min_length": 43.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.0, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9018843173980713, "rewards/meter/std": 0.12157446891069412, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9672889709472656, "rewards/repeat_soft/std": 0.04029235988855362, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.263435423374176, "rewards/total_composite/mean": 0.854701817035675, "rewards/total_composite/std": 0.09778554737567902, "reward": 0.854701817035675, "reward_std": 0.09778554737567902, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12474257498979568, "sampling/sampling_logp_difference/max": 1.9144673347473145, "sampling/importance_sampling_ratio/min": 0.14742033183574677, "sampling/importance_sampling_ratio/mean": 1.0016759634017944, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5662309415638447, "clip_ratio/low_mean": 0.05996239557862282, "clip_ratio/low_min": 0.05996239557862282, "clip_ratio/high_mean": 0.07305761240422726, "clip_ratio/high_max": 0.07305761240422726, "clip_ratio/region_mean": 0.13302000798285007, "reward_total_mean": 0.854701817035675, "reward_meter_mean": 0.9018843173980713, "reward_meter_std": 0.12157446891069412, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9672889709472656, "reward_repeat_soft_std": 0.04029235988855362, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.263435423374176, "reward_total_composite_mean": 0.854701817035675, "reward_total_composite_std": 0.09778554737567902} {"timestamp_utc": "2026-04-13T02:06:30Z", "mode": "train", "global_step": 1512, "epoch": 0.1518834756403817, "loss": 0.0166, "grad_norm": 15.132902145385742, "learning_rate": 5.421212121212122e-06, "num_tokens": 2717579.0, "completions/mean_length": 50.75, "completions/min_length": 44.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.75, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.6422211527824402, "rewards/meter/std": 0.2873256504535675, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9286820888519287, "rewards/repeat_soft/std": 0.08719930797815323, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.6799927353858948, "rewards/total_composite/std": 0.09696714580059052, "reward": 0.6799927353858948, "reward_std": 0.09696713835000992, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12478362768888474, "sampling/sampling_logp_difference/max": 1.263463020324707, "sampling/importance_sampling_ratio/min": 0.2826734185218811, "sampling/importance_sampling_ratio/mean": 1.0198009014129639, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7369678393006325, "clip_ratio/low_mean": 0.07513127010315657, "clip_ratio/low_min": 0.07513127010315657, "clip_ratio/high_mean": 0.04828748479485512, "clip_ratio/high_max": 0.04828748479485512, "clip_ratio/region_mean": 0.12341875489801168, "reward_total_mean": 0.6799927353858948, "reward_meter_mean": 0.6422211527824402, "reward_meter_std": 0.2873256504535675, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9286820888519287, "reward_repeat_soft_std": 0.08719930797815323, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.6799927353858948, "reward_total_composite_std": 0.09696714580059052} {"timestamp_utc": "2026-04-13T02:06:37Z", "mode": "train", "global_step": 1513, "epoch": 0.15198392767453542, "loss": 0.0095, "grad_norm": 9.146363258361816, "learning_rate": 5.418181818181819e-06, "num_tokens": 2719667.0, "completions/mean_length": 104.0, "completions/min_length": 85.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.0, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.9371927976608276, "rewards/meter/std": 0.16250760853290558, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8673534393310547, "rewards/repeat_soft/std": 0.11254682391881943, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7844721078872681, "rewards/total_composite/std": 0.07055628299713135, "reward": 0.7844721078872681, "reward_std": 0.07055629044771194, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11389228701591492, "sampling/sampling_logp_difference/max": 1.2606334686279297, "sampling/importance_sampling_ratio/min": 0.2834743857383728, "sampling/importance_sampling_ratio/mean": 1.006949782371521, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7536235898733139, "clip_ratio/low_mean": 0.015776699408888817, "clip_ratio/low_min": 0.015776699408888817, "clip_ratio/high_mean": 0.10119759803637862, "clip_ratio/high_max": 0.10119759803637862, "clip_ratio/region_mean": 0.11697429744526744, "reward_total_mean": 0.7844721078872681, "reward_meter_mean": 0.9371927976608276, "reward_meter_std": 0.16250760853290558, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8673534393310547, "reward_repeat_soft_std": 0.11254682391881943, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7844721078872681, "reward_total_composite_std": 0.07055628299713135} {"timestamp_utc": "2026-04-13T02:06:44Z", "mode": "train", "global_step": 1514, "epoch": 0.1520843797086891, "loss": -0.0125, "grad_norm": 9.997034072875977, "learning_rate": 5.415151515151515e-06, "num_tokens": 2721347.0, "completions/mean_length": 45.0, "completions/min_length": 40.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.0, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.9270429611206055, "rewards/meter/std": 0.13646762073040009, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9887632131576538, "rewards/repeat_soft/std": 0.009777076542377472, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7942956686019897, "rewards/total_composite/std": 0.05929591506719589, "reward": 0.7942956686019897, "reward_std": 0.059295929968357086, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14983132481575012, "sampling/sampling_logp_difference/max": 1.4595417976379395, "sampling/importance_sampling_ratio/min": 0.23234272003173828, "sampling/importance_sampling_ratio/mean": 0.9898817539215088, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8050588443875313, "clip_ratio/low_mean": 0.015625, "clip_ratio/low_min": 0.015625, "clip_ratio/high_mean": 0.1702937800437212, "clip_ratio/high_max": 0.1702937800437212, "clip_ratio/region_mean": 0.1859187800437212, "reward_total_mean": 0.7942956686019897, "reward_meter_mean": 0.9270429611206055, "reward_meter_std": 0.13646762073040009, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9887632131576538, "reward_repeat_soft_std": 0.009777076542377472, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7942956686019897, "reward_total_composite_std": 0.05929591506719589} {"timestamp_utc": "2026-04-13T02:06:53Z", "mode": "train", "global_step": 1515, "epoch": 0.15218483174284278, "loss": 0.0317, "grad_norm": 11.705793380737305, "learning_rate": 5.412121212121213e-06, "num_tokens": 2722974.0, "completions/mean_length": 36.375, "completions/min_length": 33.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.375, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.9267313480377197, "rewards/meter/std": 0.04435327649116516, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.941240668296814, "rewards/repeat_soft/std": 0.06025121733546257, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.824653148651123, "rewards/total_composite/std": 0.07683106511831284, "reward": 0.824653148651123, "reward_std": 0.07683106511831284, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08817705512046814, "sampling/sampling_logp_difference/max": 1.1541438102722168, "sampling/importance_sampling_ratio/min": 0.31532740592956543, "sampling/importance_sampling_ratio/mean": 1.0011506080627441, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43389517813920975, "clip_ratio/low_mean": 0.05475563812069595, "clip_ratio/low_min": 0.05475563812069595, "clip_ratio/high_mean": 0.02817460335791111, "clip_ratio/high_max": 0.02817460335791111, "clip_ratio/region_mean": 0.08293024147860706, "reward_total_mean": 0.824653148651123, "reward_meter_mean": 0.9267313480377197, "reward_meter_std": 0.04435327649116516, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.941240668296814, "reward_repeat_soft_std": 0.06025121733546257, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.824653148651123, "reward_total_composite_std": 0.07683106511831284} {"timestamp_utc": "2026-04-13T02:06:59Z", "mode": "train", "global_step": 1516, "epoch": 0.1522852837769965, "loss": 0.095, "grad_norm": 12.721965789794922, "learning_rate": 5.409090909090909e-06, "num_tokens": 2724488.0, "completions/mean_length": 46.25, "completions/min_length": 39.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.25, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.7273768782615662, "rewards/meter/std": 0.35448065400123596, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9447696805000305, "rewards/repeat_soft/std": 0.06679866462945938, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.19255799055099487, "rewards/total_composite/mean": 0.7135465741157532, "rewards/total_composite/std": 0.12204340845346451, "reward": 0.7135465741157532, "reward_std": 0.12204340845346451, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13141152262687683, "sampling/sampling_logp_difference/max": 1.4957075119018555, "sampling/importance_sampling_ratio/min": 0.3737737536430359, "sampling/importance_sampling_ratio/mean": 1.0252670049667358, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.923231340944767, "clip_ratio/low_mean": 0.04074883460998535, "clip_ratio/low_min": 0.04074883460998535, "clip_ratio/high_mean": 0.101004040800035, "clip_ratio/high_max": 0.101004040800035, "clip_ratio/region_mean": 0.14175287541002035, "reward_total_mean": 0.7135465741157532, "reward_meter_mean": 0.7273768782615662, "reward_meter_std": 0.35448065400123596, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9447696805000305, "reward_repeat_soft_std": 0.06679866462945938, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.19255799055099487, "reward_total_composite_mean": 0.7135465741157532, "reward_total_composite_std": 0.12204340845346451} {"timestamp_utc": "2026-04-13T02:07:06Z", "mode": "train", "global_step": 1517, "epoch": 0.15238573581115017, "loss": 0.0665, "grad_norm": 22.497989654541016, "learning_rate": 5.406060606060607e-06, "num_tokens": 2725982.0, "completions/mean_length": 26.75, "completions/min_length": 23.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.75, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9569121599197388, "rewards/meter/std": 0.03574115037918091, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9394503831863403, "rewards/repeat_soft/std": 0.05709674209356308, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8073055148124695, "rewards/total_composite/std": 0.01502560917288065, "reward": 0.8073055148124695, "reward_std": 0.015025615692138672, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14966903626918793, "sampling/sampling_logp_difference/max": 1.225515365600586, "sampling/importance_sampling_ratio/min": 0.29360634088516235, "sampling/importance_sampling_ratio/mean": 0.9902087450027466, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9559197500348091, "clip_ratio/low_mean": 0.04972448665648699, "clip_ratio/low_min": 0.04972448665648699, "clip_ratio/high_mean": 0.11569107044488192, "clip_ratio/high_max": 0.11569107044488192, "clip_ratio/region_mean": 0.1654155571013689, "reward_total_mean": 0.8073055148124695, "reward_meter_mean": 0.9569121599197388, "reward_meter_std": 0.03574115037918091, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9394503831863403, "reward_repeat_soft_std": 0.05709674209356308, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8073055148124695, "reward_total_composite_std": 0.01502560917288065} {"timestamp_utc": "2026-04-13T02:07:13Z", "mode": "train", "global_step": 1518, "epoch": 0.15248618784530388, "loss": 0.0251, "grad_norm": 6.853898048400879, "learning_rate": 5.4030303030303036e-06, "num_tokens": 2728430.0, "completions/mean_length": 114.0, "completions/min_length": 104.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.0, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.9805273413658142, "rewards/meter/std": 0.019147580489516258, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9012842178344727, "rewards/repeat_soft/std": 0.03753969445824623, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7946157455444336, "rewards/total_composite/std": 0.02163296937942505, "reward": 0.7946157455444336, "reward_std": 0.021632960066199303, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10163683444261551, "sampling/sampling_logp_difference/max": 1.518075942993164, "sampling/importance_sampling_ratio/min": 0.21913312375545502, "sampling/importance_sampling_ratio/mean": 1.0079225301742554, "sampling/importance_sampling_ratio/max": 1.9918779134750366, "entropy": 0.6408738121390343, "clip_ratio/low_mean": 0.039274530950933695, "clip_ratio/low_min": 0.039274530950933695, "clip_ratio/high_mean": 0.06201852485537529, "clip_ratio/high_max": 0.06201852485537529, "clip_ratio/region_mean": 0.10129305580630898, "reward_total_mean": 0.7946157455444336, "reward_meter_mean": 0.9805273413658142, "reward_meter_std": 0.019147580489516258, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9012842178344727, "reward_repeat_soft_std": 0.03753969445824623, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7946157455444336, "reward_total_composite_std": 0.02163296937942505} {"timestamp_utc": "2026-04-13T02:07:20Z", "mode": "train", "global_step": 1519, "epoch": 0.15258663987945756, "loss": 0.0384, "grad_norm": 9.247475624084473, "learning_rate": 5.400000000000001e-06, "num_tokens": 2730796.0, "completions/mean_length": 111.75, "completions/min_length": 105.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.75, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.9911555051803589, "rewards/meter/std": 0.002495983149856329, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8692411184310913, "rewards/repeat_soft/std": 0.10887029021978378, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7961940765380859, "rewards/total_composite/std": 0.025993645191192627, "reward": 0.7961940765380859, "reward_std": 0.025993626564741135, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10906067490577698, "sampling/sampling_logp_difference/max": 2.264272451400757, "sampling/importance_sampling_ratio/min": 0.10390559583902359, "sampling/importance_sampling_ratio/mean": 1.0070922374725342, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.676318597048521, "clip_ratio/low_mean": 0.034325397573411465, "clip_ratio/low_min": 0.034325397573411465, "clip_ratio/high_mean": 0.06751851085573435, "clip_ratio/high_max": 0.06751851085573435, "clip_ratio/region_mean": 0.10184390842914581, "reward_total_mean": 0.7961940765380859, "reward_meter_mean": 0.9911555051803589, "reward_meter_std": 0.002495983149856329, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8692411184310913, "reward_repeat_soft_std": 0.10887029021978378, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7961940765380859, "reward_total_composite_std": 0.025993645191192627} {"timestamp_utc": "2026-04-13T02:07:28Z", "mode": "train", "global_step": 1520, "epoch": 0.15268709191361124, "loss": -0.0053, "grad_norm": 8.720037460327148, "learning_rate": 5.396969696969697e-06, "num_tokens": 2733333.0, "completions/mean_length": 128.125, "completions/min_length": 117.0, "completions/max_length": 141.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.125, "completions/min_terminated_length": 117.0, "completions/max_terminated_length": 141.0, "rewards/meter/mean": 0.9719966650009155, "rewards/meter/std": 0.04096628352999687, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9434409141540527, "rewards/repeat_soft/std": 0.022438140586018562, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.6928958892822266, "rewards/total_composite/std": 0.28225967288017273, "reward": 0.6928958892822266, "reward_std": 0.28225967288017273, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11698500066995621, "sampling/sampling_logp_difference/max": 1.9939031600952148, "sampling/importance_sampling_ratio/min": 0.1361629217863083, "sampling/importance_sampling_ratio/mean": 1.0113178491592407, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6171855852007866, "clip_ratio/low_mean": 0.012605042196810246, "clip_ratio/low_min": 0.012605042196810246, "clip_ratio/high_mean": 0.09949992503970861, "clip_ratio/high_max": 0.09949992503970861, "clip_ratio/region_mean": 0.11210496723651886, "reward_total_mean": 0.6928958892822266, "reward_meter_mean": 0.9719966650009155, "reward_meter_std": 0.04096628352999687, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9434409141540527, "reward_repeat_soft_std": 0.022438140586018562, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.6928958892822266, "reward_total_composite_std": 0.28225967288017273} {"timestamp_utc": "2026-04-13T02:07:40Z", "mode": "train", "global_step": 1521, "epoch": 0.15278754394776495, "loss": -0.1593, "grad_norm": 1.671316146850586, "learning_rate": 5.3939393939393945e-06, "num_tokens": 2735046.0, "completions/mean_length": 118.125, "completions/min_length": 56.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 61.857147216796875, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.878743588924408, "rewards/meter/std": 0.3163217008113861, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9456452131271362, "rewards/repeat_soft/std": 0.03211169317364693, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.13452960550785065, "rewards/total_composite/mean": 0.7159527540206909, "rewards/total_composite/std": 0.28934445977211, "reward": 0.7159527540206909, "reward_std": 0.28934445977211, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11684088408946991, "sampling/sampling_logp_difference/max": 1.523068904876709, "sampling/importance_sampling_ratio/min": 0.21804173290729523, "sampling/importance_sampling_ratio/mean": 1.019648790359497, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7012080401182175, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09687206544913352, "clip_ratio/high_max": 0.09687206544913352, "clip_ratio/region_mean": 0.09687206544913352, "reward_total_mean": 0.7159527540206909, "reward_meter_mean": 0.878743588924408, "reward_meter_std": 0.3163217008113861, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9456452131271362, "reward_repeat_soft_std": 0.03211169317364693, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.13452960550785065, "reward_total_composite_mean": 0.7159527540206909, "reward_total_composite_std": 0.28934445977211} {"timestamp_utc": "2026-04-13T02:07:48Z", "mode": "train", "global_step": 1522, "epoch": 0.15288799598191863, "loss": 0.0212, "grad_norm": 15.037196159362793, "learning_rate": 5.390909090909091e-06, "num_tokens": 2736527.0, "completions/mean_length": 35.125, "completions/min_length": 32.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.125, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.8792842030525208, "rewards/meter/std": 0.2531192898750305, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.7362500429153442, "rewards/judge_quality/std": 0.25376805663108826, "rewards/total_composite/mean": 0.8628028631210327, "rewards/total_composite/std": 0.11109820753335953, "reward": 0.8628028631210327, "reward_std": 0.11109820008277893, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14679434895515442, "sampling/sampling_logp_difference/max": 1.3615684509277344, "sampling/importance_sampling_ratio/min": 0.2562585473060608, "sampling/importance_sampling_ratio/mean": 1.0193960666656494, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2052297666668892, "clip_ratio/low_mean": 0.09149159956723452, "clip_ratio/low_min": 0.09149159956723452, "clip_ratio/high_mean": 0.08091051131486893, "clip_ratio/high_max": 0.08091051131486893, "clip_ratio/region_mean": 0.17240211088210344, "reward_total_mean": 0.8628028631210327, "reward_meter_mean": 0.8792842030525208, "reward_meter_std": 0.2531192898750305, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.7362500429153442, "reward_judge_quality_std": 0.25376805663108826, "reward_total_composite_mean": 0.8628028631210327, "reward_total_composite_std": 0.11109820753335953} {"timestamp_utc": "2026-04-13T02:07:54Z", "mode": "train", "global_step": 1523, "epoch": 0.15298844801607234, "loss": -0.0275, "grad_norm": 9.338604927062988, "learning_rate": 5.387878787878789e-06, "num_tokens": 2738275.0, "completions/mean_length": 55.5, "completions/min_length": 44.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.5, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9935223460197449, "rewards/meter/std": 0.0022958768531680107, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9860261678695679, "rewards/repeat_soft/std": 0.019276931881904602, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8239376544952393, "rewards/total_composite/std": 0.005189390387386084, "reward": 0.8239376544952393, "reward_std": 0.0051893917843699455, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13032683730125427, "sampling/sampling_logp_difference/max": 1.7420305013656616, "sampling/importance_sampling_ratio/min": 0.1751643717288971, "sampling/importance_sampling_ratio/mean": 1.0330023765563965, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7051939256489277, "clip_ratio/low_mean": 0.09172969684004784, "clip_ratio/low_min": 0.09172969684004784, "clip_ratio/high_mean": 0.03385321516543627, "clip_ratio/high_max": 0.03385321516543627, "clip_ratio/region_mean": 0.1255829120054841, "reward_total_mean": 0.8239376544952393, "reward_meter_mean": 0.9935223460197449, "reward_meter_std": 0.0022958768531680107, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9860261678695679, "reward_repeat_soft_std": 0.019276931881904602, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8239376544952393, "reward_total_composite_std": 0.005189390387386084} {"timestamp_utc": "2026-04-13T02:08:02Z", "mode": "train", "global_step": 1524, "epoch": 0.15308890005022602, "loss": 0.0242, "grad_norm": 10.52350902557373, "learning_rate": 5.384848484848485e-06, "num_tokens": 2740461.0, "completions/mean_length": 97.25, "completions/min_length": 77.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.25, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.9024133682250977, "rewards/meter/std": 0.15376685559749603, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.859254777431488, "rewards/repeat_soft/std": 0.05565328150987625, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7569490075111389, "rewards/total_composite/std": 0.07245462387800217, "reward": 0.7569490075111389, "reward_std": 0.07245462387800217, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11561178416013718, "sampling/sampling_logp_difference/max": 3.1363677978515625, "sampling/importance_sampling_ratio/min": 0.04344029352068901, "sampling/importance_sampling_ratio/mean": 1.0037606954574585, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5090266950428486, "clip_ratio/low_mean": 0.014932647813111544, "clip_ratio/low_min": 0.014932647813111544, "clip_ratio/high_mean": 0.07307785982266068, "clip_ratio/high_max": 0.07307785982266068, "clip_ratio/region_mean": 0.08801050763577223, "reward_total_mean": 0.7569490075111389, "reward_meter_mean": 0.9024133682250977, "reward_meter_std": 0.15376685559749603, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.859254777431488, "reward_repeat_soft_std": 0.05565328150987625, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7569490075111389, "reward_total_composite_std": 0.07245462387800217} {"timestamp_utc": "2026-04-13T02:08:08Z", "mode": "train", "global_step": 1525, "epoch": 0.1531893520843797, "loss": 0.0255, "grad_norm": 11.386841773986816, "learning_rate": 5.381818181818183e-06, "num_tokens": 2742074.0, "completions/mean_length": 51.625, "completions/min_length": 48.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.625, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9875837564468384, "rewards/meter/std": 0.010906010866165161, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9123241901397705, "rewards/repeat_soft/std": 0.08658622205257416, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.18873640894889832, "rewards/total_composite/mean": 0.8438950777053833, "rewards/total_composite/std": 0.06364408135414124, "reward": 0.8438950777053833, "reward_std": 0.06364408135414124, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10502525418996811, "sampling/sampling_logp_difference/max": 1.0771112442016602, "sampling/importance_sampling_ratio/min": 0.3405779302120209, "sampling/importance_sampling_ratio/mean": 1.0198588371276855, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6346359588205814, "clip_ratio/low_mean": 0.0726764458231628, "clip_ratio/low_min": 0.0726764458231628, "clip_ratio/high_mean": 0.016741071827709675, "clip_ratio/high_max": 0.016741071827709675, "clip_ratio/region_mean": 0.08941751765087247, "reward_total_mean": 0.8438950777053833, "reward_meter_mean": 0.9875837564468384, "reward_meter_std": 0.010906010866165161, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9123241901397705, "reward_repeat_soft_std": 0.08658622205257416, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.18873640894889832, "reward_total_composite_mean": 0.8438950777053833, "reward_total_composite_std": 0.06364408135414124} {"timestamp_utc": "2026-04-13T02:08:15Z", "mode": "train", "global_step": 1526, "epoch": 0.1532898041185334, "loss": -0.0113, "grad_norm": 9.152239799499512, "learning_rate": 5.378787878787879e-06, "num_tokens": 2744123.0, "completions/mean_length": 79.125, "completions/min_length": 73.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.125, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.8105109333992004, "rewards/meter/std": 0.3271658718585968, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9720337390899658, "rewards/repeat_soft/std": 0.023902734741568565, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.7566832900047302, "rewards/total_composite/std": 0.16639196872711182, "reward": 0.7566832900047302, "reward_std": 0.16639196872711182, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12115596979856491, "sampling/sampling_logp_difference/max": 1.2765283584594727, "sampling/importance_sampling_ratio/min": 0.27900421619415283, "sampling/importance_sampling_ratio/mean": 1.0057519674301147, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7950015962123871, "clip_ratio/low_mean": 0.03858108073472977, "clip_ratio/low_min": 0.03858108073472977, "clip_ratio/high_mean": 0.08842523489147425, "clip_ratio/high_max": 0.08842523489147425, "clip_ratio/region_mean": 0.127006315626204, "reward_total_mean": 0.7566832900047302, "reward_meter_mean": 0.8105109333992004, "reward_meter_std": 0.3271658718585968, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9720337390899658, "reward_repeat_soft_std": 0.023902734741568565, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.7566832900047302, "reward_total_composite_std": 0.16639196872711182} {"timestamp_utc": "2026-04-13T02:08:22Z", "mode": "train", "global_step": 1527, "epoch": 0.1533902561526871, "loss": 0.0785, "grad_norm": 19.16898536682129, "learning_rate": 5.375757575757576e-06, "num_tokens": 2746042.0, "completions/mean_length": 76.875, "completions/min_length": 59.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.875, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.8990519046783447, "rewards/meter/std": 0.18854470551013947, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9014939665794373, "rewards/repeat_soft/std": 0.05082881823182106, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.7830977439880371, "rewards/total_composite/std": 0.09370043873786926, "reward": 0.7830977439880371, "reward_std": 0.09370043873786926, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11911214143037796, "sampling/sampling_logp_difference/max": 1.7440719604492188, "sampling/importance_sampling_ratio/min": 0.1748071312904358, "sampling/importance_sampling_ratio/mean": 0.9985024929046631, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5956007316708565, "clip_ratio/low_mean": 0.010294117964804173, "clip_ratio/low_min": 0.010294117964804173, "clip_ratio/high_mean": 0.10637703631073236, "clip_ratio/high_max": 0.10637703631073236, "clip_ratio/region_mean": 0.11667115427553654, "reward_total_mean": 0.7830977439880371, "reward_meter_mean": 0.8990519046783447, "reward_meter_std": 0.18854470551013947, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9014939665794373, "reward_repeat_soft_std": 0.05082881823182106, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.7830977439880371, "reward_total_composite_std": 0.09370043873786926} {"timestamp_utc": "2026-04-13T02:08:29Z", "mode": "train", "global_step": 1528, "epoch": 0.1534907081868408, "loss": -0.0301, "grad_norm": 7.2634687423706055, "learning_rate": 5.372727272727273e-06, "num_tokens": 2748184.0, "completions/mean_length": 96.75, "completions/min_length": 84.0, "completions/max_length": 113.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.75, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.9801443815231323, "rewards/meter/std": 0.009229694493114948, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8337709903717041, "rewards/repeat_soft/std": 0.044818215072155, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.7711920738220215, "rewards/total_composite/std": 0.03646138310432434, "reward": 0.7711920738220215, "reward_std": 0.036461394280195236, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10313063114881516, "sampling/sampling_logp_difference/max": 1.8863773345947266, "sampling/importance_sampling_ratio/min": 0.15162009000778198, "sampling/importance_sampling_ratio/mean": 1.006881833076477, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.502760972827673, "clip_ratio/low_mean": 0.055577922612428665, "clip_ratio/low_min": 0.055577922612428665, "clip_ratio/high_mean": 0.04780683107674122, "clip_ratio/high_max": 0.04780683107674122, "clip_ratio/region_mean": 0.10338475368916988, "reward_total_mean": 0.7711920738220215, "reward_meter_mean": 0.9801443815231323, "reward_meter_std": 0.009229694493114948, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8337709903717041, "reward_repeat_soft_std": 0.044818215072155, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.7711920738220215, "reward_total_composite_std": 0.03646138310432434} {"timestamp_utc": "2026-04-13T02:08:41Z", "mode": "train", "global_step": 1529, "epoch": 0.15359116022099448, "loss": -0.1997, "grad_norm": 2.352961540222168, "learning_rate": 5.36969696969697e-06, "num_tokens": 2750670.0, "completions/mean_length": 200.75, "completions/min_length": 144.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 156.2857208251953, "completions/min_terminated_length": 144.0, "completions/max_terminated_length": 172.0, "rewards/meter/mean": 0.502799928188324, "rewards/meter/std": 0.3172156512737274, "rewards/count_adherence/mean": 0.8958333134651184, "rewards/count_adherence/std": 0.294627845287323, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8066390752792358, "rewards/repeat_soft/std": 0.0995335653424263, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.5296447277069092, "rewards/total_composite/std": 0.23959745466709137, "reward": 0.5296447277069092, "reward_std": 0.23959743976593018, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10535551607608795, "sampling/sampling_logp_difference/max": 2.4333341121673584, "sampling/importance_sampling_ratio/min": 0.08774379640817642, "sampling/importance_sampling_ratio/mean": 1.0024679899215698, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5782192125916481, "clip_ratio/low_mean": 0.011217948980629444, "clip_ratio/low_min": 0.011217948980629444, "clip_ratio/high_mean": 0.0780122522264719, "clip_ratio/high_max": 0.0780122522264719, "clip_ratio/region_mean": 0.08923020120710135, "reward_total_mean": 0.5296447277069092, "reward_meter_mean": 0.502799928188324, "reward_meter_std": 0.3172156512737274, "reward_count_adherence_mean": 0.8958333134651184, "reward_count_adherence_std": 0.294627845287323, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8066390752792358, "reward_repeat_soft_std": 0.0995335653424263, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.5296447277069092, "reward_total_composite_std": 0.23959745466709137} {"timestamp_utc": "2026-04-13T02:08:49Z", "mode": "train", "global_step": 1530, "epoch": 0.15369161225514816, "loss": 0.0038, "grad_norm": 12.27077865600586, "learning_rate": 5.366666666666666e-06, "num_tokens": 2752394.0, "completions/mean_length": 53.5, "completions/min_length": 50.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.5, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.8961378931999207, "rewards/meter/std": 0.19627584517002106, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9517003297805786, "rewards/repeat_soft/std": 0.04083476960659027, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7755570411682129, "rewards/total_composite/std": 0.08807389438152313, "reward": 0.7755570411682129, "reward_std": 0.08807390183210373, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13001881539821625, "sampling/sampling_logp_difference/max": 1.8387260437011719, "sampling/importance_sampling_ratio/min": 0.15901988744735718, "sampling/importance_sampling_ratio/mean": 1.010331630706787, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7719862535595894, "clip_ratio/low_mean": 0.046509433537721634, "clip_ratio/low_min": 0.046509433537721634, "clip_ratio/high_mean": 0.08748181257396936, "clip_ratio/high_max": 0.08748181257396936, "clip_ratio/region_mean": 0.133991246111691, "reward_total_mean": 0.7755570411682129, "reward_meter_mean": 0.8961378931999207, "reward_meter_std": 0.19627584517002106, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9517003297805786, "reward_repeat_soft_std": 0.04083476960659027, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7755570411682129, "reward_total_composite_std": 0.08807389438152313} {"timestamp_utc": "2026-04-13T02:08:56Z", "mode": "train", "global_step": 1531, "epoch": 0.15379206428930187, "loss": 0.0663, "grad_norm": 16.461475372314453, "learning_rate": 5.3636363636363645e-06, "num_tokens": 2754056.0, "completions/mean_length": 57.75, "completions/min_length": 50.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.75, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.8139265775680542, "rewards/meter/std": 0.32610300183296204, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9753739833831787, "rewards/repeat_soft/std": 0.03220798820257187, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.11287637054920197, "rewards/total_composite/mean": 0.7619293928146362, "rewards/total_composite/std": 0.1413167119026184, "reward": 0.7619293928146362, "reward_std": 0.1413167268037796, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15366263687610626, "sampling/sampling_logp_difference/max": 1.526810884475708, "sampling/importance_sampling_ratio/min": 0.21722732484340668, "sampling/importance_sampling_ratio/mean": 1.0026562213897705, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8965813145041466, "clip_ratio/low_mean": 0.029873207211494446, "clip_ratio/low_min": 0.029873207211494446, "clip_ratio/high_mean": 0.11076668556779623, "clip_ratio/high_max": 0.11076668556779623, "clip_ratio/region_mean": 0.14063989277929068, "reward_total_mean": 0.7619293928146362, "reward_meter_mean": 0.8139265775680542, "reward_meter_std": 0.32610300183296204, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9753739833831787, "reward_repeat_soft_std": 0.03220798820257187, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.11287637054920197, "reward_total_composite_mean": 0.7619293928146362, "reward_total_composite_std": 0.1413167119026184} {"timestamp_utc": "2026-04-13T02:09:02Z", "mode": "train", "global_step": 1532, "epoch": 0.15389251632345555, "loss": 0.0745, "grad_norm": 14.461952209472656, "learning_rate": 5.360606060606061e-06, "num_tokens": 2755635.0, "completions/mean_length": 36.375, "completions/min_length": 31.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.375, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9209320545196533, "rewards/meter/std": 0.06621360778808594, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9817938804626465, "rewards/repeat_soft/std": 0.03823304921388626, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.25587037205696106, "rewards/total_composite/mean": 0.845973789691925, "rewards/total_composite/std": 0.07818914204835892, "reward": 0.845973789691925, "reward_std": 0.07818913459777832, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1376344859600067, "sampling/sampling_logp_difference/max": 1.7695670127868652, "sampling/importance_sampling_ratio/min": 0.1704067587852478, "sampling/importance_sampling_ratio/mean": 0.989696204662323, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6379017904400826, "clip_ratio/low_mean": 0.06305719492956996, "clip_ratio/low_min": 0.06305719492956996, "clip_ratio/high_mean": 0.05455124191939831, "clip_ratio/high_max": 0.05455124191939831, "clip_ratio/region_mean": 0.11760843684896827, "reward_total_mean": 0.845973789691925, "reward_meter_mean": 0.9209320545196533, "reward_meter_std": 0.06621360778808594, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9817938804626465, "reward_repeat_soft_std": 0.03823304921388626, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.25587037205696106, "reward_total_composite_mean": 0.845973789691925, "reward_total_composite_std": 0.07818914204835892} {"timestamp_utc": "2026-04-13T02:09:10Z", "mode": "train", "global_step": 1533, "epoch": 0.15399296835760923, "loss": -0.0484, "grad_norm": 12.308219909667969, "learning_rate": 5.357575757575758e-06, "num_tokens": 2757418.0, "completions/mean_length": 48.875, "completions/min_length": 37.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.875, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9889931082725525, "rewards/meter/std": 0.005232524126768112, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9708538055419922, "rewards/repeat_soft/std": 0.02542295679450035, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8181322813034058, "rewards/total_composite/std": 0.004391513764858246, "reward": 0.8181322813034058, "reward_std": 0.0043915072456002235, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14828188717365265, "sampling/sampling_logp_difference/max": 1.8806579113006592, "sampling/importance_sampling_ratio/min": 0.15248975157737732, "sampling/importance_sampling_ratio/mean": 1.0095511674880981, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8615017533302307, "clip_ratio/low_mean": 0.08477050624787807, "clip_ratio/low_min": 0.08477050624787807, "clip_ratio/high_mean": 0.06961753405630589, "clip_ratio/high_max": 0.06961753405630589, "clip_ratio/region_mean": 0.15438804030418396, "reward_total_mean": 0.8181322813034058, "reward_meter_mean": 0.9889931082725525, "reward_meter_std": 0.005232524126768112, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9708538055419922, "reward_repeat_soft_std": 0.02542295679450035, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8181322813034058, "reward_total_composite_std": 0.004391513764858246} {"timestamp_utc": "2026-04-13T02:09:16Z", "mode": "train", "global_step": 1534, "epoch": 0.15409342039176294, "loss": -0.0193, "grad_norm": 11.096896171569824, "learning_rate": 5.3545454545454546e-06, "num_tokens": 2759057.0, "completions/mean_length": 51.875, "completions/min_length": 45.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.875, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9748163223266602, "rewards/meter/std": 0.020872395485639572, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9757221937179565, "rewards/repeat_soft/std": 0.029674513265490532, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.08166787773370743, "rewards/total_composite/mean": 0.8006145358085632, "rewards/total_composite/std": 0.020751260221004486, "reward": 0.8006145358085632, "reward_std": 0.020751241594552994, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1458761841058731, "sampling/sampling_logp_difference/max": 1.3485186100006104, "sampling/importance_sampling_ratio/min": 0.25962457060813904, "sampling/importance_sampling_ratio/mean": 1.0133140087127686, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8828318491578102, "clip_ratio/low_mean": 0.05303983390331268, "clip_ratio/low_min": 0.05303983390331268, "clip_ratio/high_mean": 0.08484010025858879, "clip_ratio/high_max": 0.08484010025858879, "clip_ratio/region_mean": 0.13787993416190147, "reward_total_mean": 0.8006145358085632, "reward_meter_mean": 0.9748163223266602, "reward_meter_std": 0.020872395485639572, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9757221937179565, "reward_repeat_soft_std": 0.029674513265490532, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.08166787773370743, "reward_total_composite_mean": 0.8006145358085632, "reward_total_composite_std": 0.020751260221004486} {"timestamp_utc": "2026-04-13T02:09:23Z", "mode": "train", "global_step": 1535, "epoch": 0.15419387242591662, "loss": 0.0058, "grad_norm": 14.032259941101074, "learning_rate": 5.351515151515152e-06, "num_tokens": 2760705.0, "completions/mean_length": 48.0, "completions/min_length": 45.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.0, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9364473223686218, "rewards/meter/std": 0.09013306349515915, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9613661766052246, "rewards/repeat_soft/std": 0.034090153872966766, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7957879304885864, "rewards/total_composite/std": 0.03804502263665199, "reward": 0.7957879304885864, "reward_std": 0.038045018911361694, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11854260414838791, "sampling/sampling_logp_difference/max": 2.3604073524475098, "sampling/importance_sampling_ratio/min": 0.09438176453113556, "sampling/importance_sampling_ratio/mean": 0.9982342720031738, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5632080510258675, "clip_ratio/low_mean": 0.03191489353775978, "clip_ratio/low_min": 0.03191489353775978, "clip_ratio/high_mean": 0.08424328081309795, "clip_ratio/high_max": 0.08424328081309795, "clip_ratio/region_mean": 0.11615817435085773, "reward_total_mean": 0.7957879304885864, "reward_meter_mean": 0.9364473223686218, "reward_meter_std": 0.09013306349515915, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9613661766052246, "reward_repeat_soft_std": 0.034090153872966766, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7957879304885864, "reward_total_composite_std": 0.03804502263665199} {"timestamp_utc": "2026-04-13T02:09:36Z", "mode": "train", "global_step": 1536, "epoch": 0.15429432446007033, "loss": -0.161, "grad_norm": 4.028048992156982, "learning_rate": 5.348484848484848e-06, "num_tokens": 2762790.0, "completions/mean_length": 152.625, "completions/min_length": 92.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 101.28572082519531, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.6155634522438049, "rewards/meter/std": 0.4058144986629486, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9647611379623413, "rewards/repeat_soft/std": 0.044223278760910034, "rewards/judge_quality/mean": 0.29750001430511475, "rewards/judge_quality/std": 0.14518460631370544, "rewards/total_composite/mean": 0.5986671447753906, "rewards/total_composite/std": 0.23811818659305573, "reward": 0.5986671447753906, "reward_std": 0.23811818659305573, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1438247561454773, "sampling/sampling_logp_difference/max": 2.5389657020568848, "sampling/importance_sampling_ratio/min": 0.078948013484478, "sampling/importance_sampling_ratio/mean": 1.0029011964797974, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6881646066904068, "clip_ratio/low_mean": 0.03862664848566055, "clip_ratio/low_min": 0.03862664848566055, "clip_ratio/high_mean": 0.07260556519031525, "clip_ratio/high_max": 0.07260556519031525, "clip_ratio/region_mean": 0.1112322136759758, "reward_total_mean": 0.5986671447753906, "reward_meter_mean": 0.6155634522438049, "reward_meter_std": 0.4058144986629486, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.2651650309562683, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9647611379623413, "reward_repeat_soft_std": 0.044223278760910034, "reward_judge_quality_mean": 0.29750001430511475, "reward_judge_quality_std": 0.14518460631370544, "reward_total_composite_mean": 0.5986671447753906, "reward_total_composite_std": 0.23811818659305573} {"timestamp_utc": "2026-04-13T02:09:43Z", "mode": "train", "global_step": 1537, "epoch": 0.154394776494224, "loss": 0.03, "grad_norm": 7.127829551696777, "learning_rate": 5.3454545454545455e-06, "num_tokens": 2765701.0, "completions/mean_length": 174.875, "completions/min_length": 167.0, "completions/max_length": 180.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 174.875, "completions/min_terminated_length": 167.0, "completions/max_terminated_length": 180.0, "rewards/meter/mean": 0.9821039438247681, "rewards/meter/std": 0.017306072637438774, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.786784291267395, "rewards/repeat_soft/std": 0.12661457061767578, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.7711251974105835, "rewards/total_composite/std": 0.02846049703657627, "reward": 0.7711251974105835, "reward_std": 0.028460495173931122, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09635216742753983, "sampling/sampling_logp_difference/max": 1.8673210144042969, "sampling/importance_sampling_ratio/min": 0.1545371115207672, "sampling/importance_sampling_ratio/mean": 1.0096073150634766, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5879133567214012, "clip_ratio/low_mean": 0.039578068535774946, "clip_ratio/low_min": 0.039578068535774946, "clip_ratio/high_mean": 0.04481323901563883, "clip_ratio/high_max": 0.04481323901563883, "clip_ratio/region_mean": 0.08439130755141377, "reward_total_mean": 0.7711251974105835, "reward_meter_mean": 0.9821039438247681, "reward_meter_std": 0.017306072637438774, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.786784291267395, "reward_repeat_soft_std": 0.12661457061767578, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.7711251974105835, "reward_total_composite_std": 0.02846049703657627} {"timestamp_utc": "2026-04-13T02:09:51Z", "mode": "train", "global_step": 1538, "epoch": 0.1544952285283777, "loss": 0.0032, "grad_norm": 7.256498336791992, "learning_rate": 5.342424242424244e-06, "num_tokens": 2767998.0, "completions/mean_length": 102.125, "completions/min_length": 91.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.125, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.9921061992645264, "rewards/meter/std": 0.0038761168252676725, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8062028884887695, "rewards/repeat_soft/std": 0.0744187980890274, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7856305837631226, "rewards/total_composite/std": 0.02601303905248642, "reward": 0.7856305837631226, "reward_std": 0.026013029739260674, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08667322993278503, "sampling/sampling_logp_difference/max": 1.3694062232971191, "sampling/importance_sampling_ratio/min": 0.2542578876018524, "sampling/importance_sampling_ratio/mean": 1.005146861076355, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4606965370476246, "clip_ratio/low_mean": 0.03619278408586979, "clip_ratio/low_min": 0.03619278408586979, "clip_ratio/high_mean": 0.053524527698755264, "clip_ratio/high_max": 0.053524527698755264, "clip_ratio/region_mean": 0.08971731178462505, "reward_total_mean": 0.7856305837631226, "reward_meter_mean": 0.9921061992645264, "reward_meter_std": 0.0038761168252676725, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8062028884887695, "reward_repeat_soft_std": 0.0744187980890274, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7856305837631226, "reward_total_composite_std": 0.02601303905248642} {"timestamp_utc": "2026-04-13T02:09:59Z", "mode": "train", "global_step": 1539, "epoch": 0.1545956805625314, "loss": 0.0408, "grad_norm": 12.574394226074219, "learning_rate": 5.33939393939394e-06, "num_tokens": 2769717.0, "completions/mean_length": 54.875, "completions/min_length": 51.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.875, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.8657891750335693, "rewards/meter/std": 0.3011331558227539, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9463545083999634, "rewards/repeat_soft/std": 0.051718007773160934, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.263435423374176, "rewards/total_composite/mean": 0.8363655805587769, "rewards/total_composite/std": 0.182621568441391, "reward": 0.8363655805587769, "reward_std": 0.182621568441391, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1348583996295929, "sampling/sampling_logp_difference/max": 1.3223991394042969, "sampling/importance_sampling_ratio/min": 0.2664951980113983, "sampling/importance_sampling_ratio/mean": 1.0010473728179932, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7971316874027252, "clip_ratio/low_mean": 0.07512038387358189, "clip_ratio/low_min": 0.07512038387358189, "clip_ratio/high_mean": 0.07130359392613173, "clip_ratio/high_max": 0.07130359392613173, "clip_ratio/region_mean": 0.1464239777997136, "reward_total_mean": 0.8363655805587769, "reward_meter_mean": 0.8657891750335693, "reward_meter_std": 0.3011331558227539, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9463545083999634, "reward_repeat_soft_std": 0.051718007773160934, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.263435423374176, "reward_total_composite_mean": 0.8363655805587769, "reward_total_composite_std": 0.182621568441391} {"timestamp_utc": "2026-04-13T02:10:07Z", "mode": "train", "global_step": 1540, "epoch": 0.15469613259668508, "loss": -0.0059, "grad_norm": 7.003292560577393, "learning_rate": 5.336363636363637e-06, "num_tokens": 2772022.0, "completions/mean_length": 119.125, "completions/min_length": 111.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.125, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.9742308259010315, "rewards/meter/std": 0.029152411967515945, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9333407878875732, "rewards/repeat_soft/std": 0.05171765759587288, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.14520922303199768, "rewards/total_composite/mean": 0.8062379360198975, "rewards/total_composite/std": 0.04609382897615433, "reward": 0.8062379360198975, "reward_std": 0.04609382152557373, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11294770985841751, "sampling/sampling_logp_difference/max": 2.8070387840270996, "sampling/importance_sampling_ratio/min": 0.06038353592157364, "sampling/importance_sampling_ratio/mean": 1.01071298122406, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6918713599443436, "clip_ratio/low_mean": 0.04100188426673412, "clip_ratio/low_min": 0.04100188426673412, "clip_ratio/high_mean": 0.07622228097170591, "clip_ratio/high_max": 0.07622228097170591, "clip_ratio/region_mean": 0.11722416523844004, "reward_total_mean": 0.8062379360198975, "reward_meter_mean": 0.9742308259010315, "reward_meter_std": 0.029152411967515945, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9333407878875732, "reward_repeat_soft_std": 0.05171765759587288, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.14520922303199768, "reward_total_composite_mean": 0.8062379360198975, "reward_total_composite_std": 0.04609382897615433} {"timestamp_utc": "2026-04-13T02:10:14Z", "mode": "train", "global_step": 1541, "epoch": 0.15479658463083878, "loss": 0.0127, "grad_norm": 9.64117431640625, "learning_rate": 5.333333333333334e-06, "num_tokens": 2774279.0, "completions/mean_length": 94.125, "completions/min_length": 84.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.125, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.9935216903686523, "rewards/meter/std": 0.0018256056355312467, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8713598251342773, "rewards/repeat_soft/std": 0.0701516717672348, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.8289707899093628, "rewards/total_composite/std": 0.04961363226175308, "reward": 0.8289707899093628, "reward_std": 0.04961363226175308, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11288685351610184, "sampling/sampling_logp_difference/max": 2.1479461193084717, "sampling/importance_sampling_ratio/min": 0.11672364175319672, "sampling/importance_sampling_ratio/mean": 1.0043526887893677, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4943540319800377, "clip_ratio/low_mean": 0.09664009511470795, "clip_ratio/low_min": 0.09664009511470795, "clip_ratio/high_mean": 0.009408602491021156, "clip_ratio/high_max": 0.009408602491021156, "clip_ratio/region_mean": 0.1060486976057291, "reward_total_mean": 0.8289707899093628, "reward_meter_mean": 0.9935216903686523, "reward_meter_std": 0.0018256056355312467, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8713598251342773, "reward_repeat_soft_std": 0.0701516717672348, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.8289707899093628, "reward_total_composite_std": 0.04961363226175308} {"timestamp_utc": "2026-04-13T02:10:21Z", "mode": "train", "global_step": 1542, "epoch": 0.15489703666499247, "loss": 0.0199, "grad_norm": 14.714439392089844, "learning_rate": 5.330303030303031e-06, "num_tokens": 2775811.0, "completions/mean_length": 48.5, "completions/min_length": 44.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.5, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9452280402183533, "rewards/meter/std": 0.07617586851119995, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9684310555458069, "rewards/repeat_soft/std": 0.03588847443461418, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.8731957077980042, "rewards/total_composite/std": 0.07239528745412827, "reward": 0.8731957077980042, "reward_std": 0.07239526510238647, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1351664960384369, "sampling/sampling_logp_difference/max": 1.8007078170776367, "sampling/importance_sampling_ratio/min": 0.1651819348335266, "sampling/importance_sampling_ratio/mean": 0.9931858777999878, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7241077274084091, "clip_ratio/low_mean": 0.0473348880186677, "clip_ratio/low_min": 0.0473348880186677, "clip_ratio/high_mean": 0.0745833357796073, "clip_ratio/high_max": 0.0745833357796073, "clip_ratio/region_mean": 0.121918223798275, "reward_total_mean": 0.8731957077980042, "reward_meter_mean": 0.9452280402183533, "reward_meter_std": 0.07617586851119995, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9684310555458069, "reward_repeat_soft_std": 0.03588847443461418, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.8731957077980042, "reward_total_composite_std": 0.07239528745412827} {"timestamp_utc": "2026-04-13T02:10:29Z", "mode": "train", "global_step": 1543, "epoch": 0.15499748869914615, "loss": -0.0129, "grad_norm": 6.2471418380737305, "learning_rate": 5.327272727272727e-06, "num_tokens": 2778471.0, "completions/mean_length": 146.5, "completions/min_length": 128.0, "completions/max_length": 174.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 146.5, "completions/min_terminated_length": 128.0, "completions/max_terminated_length": 174.0, "rewards/meter/mean": 0.9880247116088867, "rewards/meter/std": 0.008001437410712242, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.08625820279121399, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.671135663986206, "rewards/repeat_soft/std": 0.07135070115327835, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7592246532440186, "rewards/total_composite/std": 0.03175835683941841, "reward": 0.7592246532440186, "reward_std": 0.031758371740579605, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09886561334133148, "sampling/sampling_logp_difference/max": 5.407557487487793, "sampling/importance_sampling_ratio/min": 0.0044825756922364235, "sampling/importance_sampling_ratio/mean": 0.9957990646362305, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4791370555758476, "clip_ratio/low_mean": 0.02865055133588612, "clip_ratio/low_min": 0.02865055133588612, "clip_ratio/high_mean": 0.051499187014997005, "clip_ratio/high_max": 0.051499187014997005, "clip_ratio/region_mean": 0.08014973835088313, "reward_total_mean": 0.7592246532440186, "reward_meter_mean": 0.9880247116088867, "reward_meter_std": 0.008001437410712242, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.08625820279121399, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.671135663986206, "reward_repeat_soft_std": 0.07135070115327835, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7592246532440186, "reward_total_composite_std": 0.03175835683941841} {"timestamp_utc": "2026-04-13T02:10:36Z", "mode": "train", "global_step": 1544, "epoch": 0.15509794073329985, "loss": 0.0357, "grad_norm": 9.24489974975586, "learning_rate": 5.324242424242425e-06, "num_tokens": 2780360.0, "completions/mean_length": 61.125, "completions/min_length": 57.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.125, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9897646903991699, "rewards/meter/std": 0.004917032551020384, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9586600661277771, "rewards/repeat_soft/std": 0.03699398413300514, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740273475647, "rewards/total_composite/mean": 0.8382601141929626, "rewards/total_composite/std": 0.05078065022826195, "reward": 0.8382601141929626, "reward_std": 0.05078066140413284, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11233915388584137, "sampling/sampling_logp_difference/max": 1.6410503387451172, "sampling/importance_sampling_ratio/min": 0.19377639889717102, "sampling/importance_sampling_ratio/mean": 1.0026445388793945, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7441268637776375, "clip_ratio/low_mean": 0.09634613012894988, "clip_ratio/low_min": 0.09634613012894988, "clip_ratio/high_mean": 0.006355931982398033, "clip_ratio/high_max": 0.006355931982398033, "clip_ratio/region_mean": 0.10270206211134791, "reward_total_mean": 0.8382601141929626, "reward_meter_mean": 0.9897646903991699, "reward_meter_std": 0.004917032551020384, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9586600661277771, "reward_repeat_soft_std": 0.03699398413300514, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740273475647, "reward_total_composite_mean": 0.8382601141929626, "reward_total_composite_std": 0.05078065022826195} {"timestamp_utc": "2026-04-13T02:10:43Z", "mode": "train", "global_step": 1545, "epoch": 0.15519839276745354, "loss": 0.0835, "grad_norm": 11.194290161132812, "learning_rate": 5.321212121212122e-06, "num_tokens": 2782116.0, "completions/mean_length": 50.5, "completions/min_length": 42.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.5, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9869616031646729, "rewards/meter/std": 0.012553077191114426, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9565790891647339, "rewards/repeat_soft/std": 0.03190453723073006, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.8367906212806702, "rewards/total_composite/std": 0.05448918417096138, "reward": 0.8367906212806702, "reward_std": 0.054489172995090485, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13033612072467804, "sampling/sampling_logp_difference/max": 3.198460340499878, "sampling/importance_sampling_ratio/min": 0.0408250130712986, "sampling/importance_sampling_ratio/mean": 1.0042649507522583, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7201526835560799, "clip_ratio/low_mean": 0.1162795927375555, "clip_ratio/low_min": 0.1162795927375555, "clip_ratio/high_mean": 0.02678571455180645, "clip_ratio/high_max": 0.02678571455180645, "clip_ratio/region_mean": 0.14306530728936195, "reward_total_mean": 0.8367906212806702, "reward_meter_mean": 0.9869616031646729, "reward_meter_std": 0.012553077191114426, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9565790891647339, "reward_repeat_soft_std": 0.03190453723073006, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.8367906212806702, "reward_total_composite_std": 0.05448918417096138} {"timestamp_utc": "2026-04-13T02:10:51Z", "mode": "train", "global_step": 1546, "epoch": 0.15529884480160724, "loss": -0.0378, "grad_norm": 9.930231094360352, "learning_rate": 5.318181818181819e-06, "num_tokens": 2784211.0, "completions/mean_length": 102.875, "completions/min_length": 75.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.875, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.9850201606750488, "rewards/meter/std": 0.02198101207613945, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8165416717529297, "rewards/repeat_soft/std": 0.07340390980243683, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8009132146835327, "rewards/total_composite/std": 0.006869843695312738, "reward": 0.8009132146835327, "reward_std": 0.006869844626635313, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12489032000303268, "sampling/sampling_logp_difference/max": 1.8908672332763672, "sampling/importance_sampling_ratio/min": 0.15094085037708282, "sampling/importance_sampling_ratio/mean": 1.0123053789138794, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7039178088307381, "clip_ratio/low_mean": 0.06003732793033123, "clip_ratio/low_min": 0.06003732793033123, "clip_ratio/high_mean": 0.059355515986680984, "clip_ratio/high_max": 0.059355515986680984, "clip_ratio/region_mean": 0.11939284391701221, "reward_total_mean": 0.8009132146835327, "reward_meter_mean": 0.9850201606750488, "reward_meter_std": 0.02198101207613945, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8165416717529297, "reward_repeat_soft_std": 0.07340390980243683, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8009132146835327, "reward_total_composite_std": 0.006869843695312738} {"timestamp_utc": "2026-04-13T02:10:58Z", "mode": "train", "global_step": 1547, "epoch": 0.15539929683576093, "loss": 0.0547, "grad_norm": 8.518730163574219, "learning_rate": 5.3151515151515155e-06, "num_tokens": 2786228.0, "completions/mean_length": 77.125, "completions/min_length": 69.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.125, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.7297423481941223, "rewards/meter/std": 0.2607722282409668, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9089797139167786, "rewards/repeat_soft/std": 0.06987927854061127, "rewards/judge_quality/mean": 0.6487500071525574, "rewards/judge_quality/std": 0.14156651496887207, "rewards/total_composite/mean": 0.7639070153236389, "rewards/total_composite/std": 0.11812534928321838, "reward": 0.7639070153236389, "reward_std": 0.11812533438205719, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11298324167728424, "sampling/sampling_logp_difference/max": 1.6958153247833252, "sampling/importance_sampling_ratio/min": 0.18344959616661072, "sampling/importance_sampling_ratio/mean": 0.993315577507019, "sampling/importance_sampling_ratio/max": 1.9852479696273804, "entropy": 0.4432750791311264, "clip_ratio/low_mean": 0.03676997032016516, "clip_ratio/low_min": 0.03676997032016516, "clip_ratio/high_mean": 0.06098391441628337, "clip_ratio/high_max": 0.06098391441628337, "clip_ratio/region_mean": 0.09775388473644853, "reward_total_mean": 0.7639070153236389, "reward_meter_mean": 0.7297423481941223, "reward_meter_std": 0.2607722282409668, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9089797139167786, "reward_repeat_soft_std": 0.06987927854061127, "reward_judge_quality_mean": 0.6487500071525574, "reward_judge_quality_std": 0.14156651496887207, "reward_total_composite_mean": 0.7639070153236389, "reward_total_composite_std": 0.11812534928321838} {"timestamp_utc": "2026-04-13T02:11:05Z", "mode": "train", "global_step": 1548, "epoch": 0.1554997488699146, "loss": 0.0359, "grad_norm": 8.19249153137207, "learning_rate": 5.312121212121213e-06, "num_tokens": 2788673.0, "completions/mean_length": 114.625, "completions/min_length": 107.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.625, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.9634779691696167, "rewards/meter/std": 0.07320486009120941, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9543237090110779, "rewards/repeat_soft/std": 0.028520595282316208, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.8115599155426025, "rewards/total_composite/std": 0.05073153227567673, "reward": 0.8115599155426025, "reward_std": 0.050731539726257324, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13271592557430267, "sampling/sampling_logp_difference/max": 3.3692121505737305, "sampling/importance_sampling_ratio/min": 0.034416742622852325, "sampling/importance_sampling_ratio/mean": 1.0023924112319946, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6279976442456245, "clip_ratio/low_mean": 0.0471173832193017, "clip_ratio/low_min": 0.0471173832193017, "clip_ratio/high_mean": 0.09257827512919903, "clip_ratio/high_max": 0.09257827512919903, "clip_ratio/region_mean": 0.13969565834850073, "reward_total_mean": 0.8115599155426025, "reward_meter_mean": 0.9634779691696167, "reward_meter_std": 0.07320486009120941, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9543237090110779, "reward_repeat_soft_std": 0.028520595282316208, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.8115599155426025, "reward_total_composite_std": 0.05073153227567673} {"timestamp_utc": "2026-04-13T02:11:12Z", "mode": "train", "global_step": 1549, "epoch": 0.15560020090406831, "loss": 0.0572, "grad_norm": 22.67731285095215, "learning_rate": 5.309090909090909e-06, "num_tokens": 2790313.0, "completions/mean_length": 34.0, "completions/min_length": 30.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.0, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.70359867811203, "rewards/meter/std": 0.27570971846580505, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9274407625198364, "rewards/repeat_soft/std": 0.0597173273563385, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.7978634834289551, "rewards/total_composite/std": 0.17846746742725372, "reward": 0.7978634834289551, "reward_std": 0.17846745252609253, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15657006204128265, "sampling/sampling_logp_difference/max": 2.1951117515563965, "sampling/importance_sampling_ratio/min": 0.11134611815214157, "sampling/importance_sampling_ratio/mean": 1.0003782510757446, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5615999847650528, "clip_ratio/low_mean": 0.07919884286820889, "clip_ratio/low_min": 0.07919884286820889, "clip_ratio/high_mean": 0.08878577221184969, "clip_ratio/high_max": 0.08878577221184969, "clip_ratio/region_mean": 0.16798461508005857, "reward_total_mean": 0.7978634834289551, "reward_meter_mean": 0.70359867811203, "reward_meter_std": 0.27570971846580505, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9274407625198364, "reward_repeat_soft_std": 0.0597173273563385, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.7978634834289551, "reward_total_composite_std": 0.17846746742725372} {"timestamp_utc": "2026-04-13T02:11:19Z", "mode": "train", "global_step": 1550, "epoch": 0.155700652938222, "loss": -0.0325, "grad_norm": 18.983745574951172, "learning_rate": 5.306060606060606e-06, "num_tokens": 2791832.0, "completions/mean_length": 26.875, "completions/min_length": 23.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.875, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.8941363096237183, "rewards/meter/std": 0.21150271594524384, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9390609264373779, "rewards/repeat_soft/std": 0.0598461776971817, "rewards/judge_quality/mean": 0.3725000023841858, "rewards/judge_quality/std": 0.11055056750774384, "rewards/total_composite/mean": 0.7580174207687378, "rewards/total_composite/std": 0.08958832174539566, "reward": 0.7580174207687378, "reward_std": 0.08958834409713745, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13754114508628845, "sampling/sampling_logp_difference/max": 1.365687370300293, "sampling/importance_sampling_ratio/min": 0.2552051842212677, "sampling/importance_sampling_ratio/mean": 1.0149439573287964, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6715265847742558, "clip_ratio/low_mean": 0.03035982046276331, "clip_ratio/low_min": 0.03035982046276331, "clip_ratio/high_mean": 0.09863068908452988, "clip_ratio/high_max": 0.09863068908452988, "clip_ratio/region_mean": 0.1289905095472932, "reward_total_mean": 0.7580174207687378, "reward_meter_mean": 0.8941363096237183, "reward_meter_std": 0.21150271594524384, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9390609264373779, "reward_repeat_soft_std": 0.0598461776971817, "reward_judge_quality_mean": 0.3725000023841858, "reward_judge_quality_std": 0.11055056750774384, "reward_total_composite_mean": 0.7580174207687378, "reward_total_composite_std": 0.08958832174539566} {"timestamp_utc": "2026-04-13T02:12:13Z", "mode": "eval", "global_step": 1550, "epoch": 0.155700652938222, "eval_loss": NaN, "eval_runtime": 54.0266, "eval_samples_per_second": 1.481, "eval_steps_per_second": 0.185, "eval_num_tokens": 2791832.0, "eval_completions/mean_length": 85.325, "eval_completions/min_length": 37.3, "eval_completions/max_length": 143.9, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 85.325, "eval_completions/min_terminated_length": 37.3, "eval_completions/max_terminated_length": 143.9, "eval_rewards/meter/mean": 0.856160581111908, "eval_rewards/meter/std": 0.23412120938301087, "eval_rewards/count_adherence/mean": 1.0, "eval_rewards/count_adherence/std": 0.0, "eval_rewards/hard_gate/mean": 0.9625, "eval_rewards/hard_gate/std": 0.10606601536273956, "eval_rewards/repeat_soft/mean": 0.9263713419437408, "eval_rewards/repeat_soft/std": 0.07464694194495677, "eval_rewards/judge_quality/mean": 0.42299999892711637, "eval_rewards/judge_quality/std": 0.11201798841357231, "eval_rewards/total_composite/mean": 0.7343337178230286, "eval_rewards/total_composite/std": 0.1724328838288784, "eval_reward": 0.7343337178230286, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.055419334396719935, "eval_sampling/sampling_logp_difference/max": 0.9716408729553223, "eval_sampling/importance_sampling_ratio/min": 0.38728341460227966, "eval_sampling/importance_sampling_ratio/mean": 1.0133357048034668, "eval_sampling/importance_sampling_ratio/max": 1.4820619106292725, "eval_entropy": 0.6227017015218734, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7343337178230286, "eval_reward_meter_mean": 0.856160581111908, "eval_reward_meter_std": 0.23412120938301087, "eval_reward_count_adherence_mean": 1.0, "eval_reward_count_adherence_std": 0.0, "eval_reward_hard_gate_mean": 0.9625, "eval_reward_hard_gate_std": 0.10606601536273956, "eval_reward_repeat_soft_mean": 0.9263713419437408, "eval_reward_repeat_soft_std": 0.07464694194495677, "eval_reward_judge_quality_mean": 0.42299999892711637, "eval_reward_judge_quality_std": 0.11201798841357231, "eval_reward_total_composite_mean": 0.7343337178230286, "eval_reward_total_composite_std": 0.1724328838288784} {"timestamp_utc": "2026-04-13T02:12:25Z", "mode": "train", "global_step": 1551, "epoch": 0.1558011049723757, "loss": 0.0318, "grad_norm": 7.148831367492676, "learning_rate": 5.303030303030303e-06, "num_tokens": 2794773.0, "completions/mean_length": 159.625, "completions/min_length": 147.0, "completions/max_length": 183.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 159.625, "completions/min_terminated_length": 147.0, "completions/max_terminated_length": 183.0, "rewards/meter/mean": 0.9899540543556213, "rewards/meter/std": 0.006341850850731134, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7628903388977051, "rewards/repeat_soft/std": 0.08792707324028015, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.7722683548927307, "rewards/total_composite/std": 0.030029507353901863, "reward": 0.7722683548927307, "reward_std": 0.030029485002160072, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11344742029905319, "sampling/sampling_logp_difference/max": 4.0734758377075195, "sampling/importance_sampling_ratio/min": 0.017018131911754608, "sampling/importance_sampling_ratio/mean": 1.001677393913269, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.519325751811266, "clip_ratio/low_mean": 0.05138283595442772, "clip_ratio/low_min": 0.05138283595442772, "clip_ratio/high_mean": 0.0545952832326293, "clip_ratio/high_max": 0.0545952832326293, "clip_ratio/region_mean": 0.10597811918705702, "reward_total_mean": 0.7722683548927307, "reward_meter_mean": 0.9899540543556213, "reward_meter_std": 0.006341850850731134, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7628903388977051, "reward_repeat_soft_std": 0.08792707324028015, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.7722683548927307, "reward_total_composite_std": 0.030029507353901863} {"timestamp_utc": "2026-04-13T02:12:31Z", "mode": "train", "global_step": 1552, "epoch": 0.15590155700652938, "loss": 0.0672, "grad_norm": 16.063390731811523, "learning_rate": 5.300000000000001e-06, "num_tokens": 2796199.0, "completions/mean_length": 25.25, "completions/min_length": 22.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.25, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9884547591209412, "rewards/meter/std": 0.008594991639256477, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9539694786071777, "rewards/repeat_soft/std": 0.011207234114408493, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.19334925711154938, "rewards/total_composite/mean": 0.8308265209197998, "rewards/total_composite/std": 0.0584600530564785, "reward": 0.8308265209197998, "reward_std": 0.05846007168292999, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13370336592197418, "sampling/sampling_logp_difference/max": 3.727304697036743, "sampling/importance_sampling_ratio/min": 0.02405759133398533, "sampling/importance_sampling_ratio/mean": 1.0009992122650146, "sampling/importance_sampling_ratio/max": 1.5508718490600586, "entropy": 0.5975765213370323, "clip_ratio/low_mean": 0.11149653745815158, "clip_ratio/low_min": 0.11149653745815158, "clip_ratio/high_mean": 0.011363636702299118, "clip_ratio/high_max": 0.011363636702299118, "clip_ratio/region_mean": 0.1228601741604507, "reward_total_mean": 0.8308265209197998, "reward_meter_mean": 0.9884547591209412, "reward_meter_std": 0.008594991639256477, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9539694786071777, "reward_repeat_soft_std": 0.011207234114408493, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.19334925711154938, "reward_total_composite_mean": 0.8308265209197998, "reward_total_composite_std": 0.0584600530564785} {"timestamp_utc": "2026-04-13T02:12:38Z", "mode": "train", "global_step": 1553, "epoch": 0.15600200904068307, "loss": 0.0855, "grad_norm": 14.639063835144043, "learning_rate": 5.296969696969697e-06, "num_tokens": 2797791.0, "completions/mean_length": 49.0, "completions/min_length": 45.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.0, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9783709049224854, "rewards/meter/std": 0.013834037818014622, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9973539710044861, "rewards/repeat_soft/std": 0.0029955666977912188, "rewards/judge_quality/mean": 0.8575000166893005, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.9472523331642151, "rewards/total_composite/std": 0.05148541182279587, "reward": 0.9472523331642151, "reward_std": 0.05148540064692497, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1259874403476715, "sampling/sampling_logp_difference/max": 1.6004894971847534, "sampling/importance_sampling_ratio/min": 0.2017977088689804, "sampling/importance_sampling_ratio/mean": 0.9933533668518066, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6531937196850777, "clip_ratio/low_mean": 0.019736841320991516, "clip_ratio/low_min": 0.019736841320991516, "clip_ratio/high_mean": 0.09103939821943641, "clip_ratio/high_max": 0.09103939821943641, "clip_ratio/region_mean": 0.11077623954042792, "reward_total_mean": 0.9472523331642151, "reward_meter_mean": 0.9783709049224854, "reward_meter_std": 0.013834037818014622, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9973539710044861, "reward_repeat_soft_std": 0.0029955666977912188, "reward_judge_quality_mean": 0.8575000166893005, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.9472523331642151, "reward_total_composite_std": 0.05148541182279587} {"timestamp_utc": "2026-04-13T02:12:45Z", "mode": "train", "global_step": 1554, "epoch": 0.15610246107483677, "loss": 0.123, "grad_norm": 22.95470428466797, "learning_rate": 5.293939393939395e-06, "num_tokens": 2799156.0, "completions/mean_length": 26.625, "completions/min_length": 21.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.625, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.8701572418212891, "rewards/meter/std": 0.2083621472120285, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9575977325439453, "rewards/repeat_soft/std": 0.013865554705262184, "rewards/judge_quality/mean": 0.7987500429153442, "rewards/judge_quality/std": 0.22465452551841736, "rewards/total_composite/mean": 0.876955509185791, "rewards/total_composite/std": 0.09723363071680069, "reward": 0.876955509185791, "reward_std": 0.0972336158156395, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1495901197195053, "sampling/sampling_logp_difference/max": 1.1541123390197754, "sampling/importance_sampling_ratio/min": 0.31533733010292053, "sampling/importance_sampling_ratio/mean": 1.0275248289108276, "sampling/importance_sampling_ratio/max": 1.9591610431671143, "entropy": 0.798930186778307, "clip_ratio/low_mean": 0.08169229701161385, "clip_ratio/low_min": 0.08169229701161385, "clip_ratio/high_mean": 0.09785042703151703, "clip_ratio/high_max": 0.09785042703151703, "clip_ratio/region_mean": 0.17954272404313087, "reward_total_mean": 0.876955509185791, "reward_meter_mean": 0.8701572418212891, "reward_meter_std": 0.2083621472120285, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9575977325439453, "reward_repeat_soft_std": 0.013865554705262184, "reward_judge_quality_mean": 0.7987500429153442, "reward_judge_quality_std": 0.22465452551841736, "reward_total_composite_mean": 0.876955509185791, "reward_total_composite_std": 0.09723363071680069} {"timestamp_utc": "2026-04-13T02:12:53Z", "mode": "train", "global_step": 1555, "epoch": 0.15620291310899045, "loss": 0.0732, "grad_norm": 8.537090301513672, "learning_rate": 5.290909090909091e-06, "num_tokens": 2801088.0, "completions/mean_length": 75.5, "completions/min_length": 66.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.5, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.9123576283454895, "rewards/meter/std": 0.05767492949962616, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9593722224235535, "rewards/repeat_soft/std": 0.04451820254325867, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.8012481331825256, "rewards/total_composite/std": 0.06636017560958862, "reward": 0.8012481331825256, "reward_std": 0.06636016815900803, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09272019565105438, "sampling/sampling_logp_difference/max": 1.9389114379882812, "sampling/importance_sampling_ratio/min": 0.14386045932769775, "sampling/importance_sampling_ratio/mean": 1.0091032981872559, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43078984692692757, "clip_ratio/low_mean": 0.06543680001050234, "clip_ratio/low_min": 0.06543680001050234, "clip_ratio/high_mean": 0.03165584499947727, "clip_ratio/high_max": 0.03165584499947727, "clip_ratio/region_mean": 0.0970926450099796, "reward_total_mean": 0.8012481331825256, "reward_meter_mean": 0.9123576283454895, "reward_meter_std": 0.05767492949962616, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9593722224235535, "reward_repeat_soft_std": 0.04451820254325867, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.8012481331825256, "reward_total_composite_std": 0.06636017560958862} {"timestamp_utc": "2026-04-13T02:13:01Z", "mode": "train", "global_step": 1556, "epoch": 0.15630336514314414, "loss": 0.0569, "grad_norm": 11.210988998413086, "learning_rate": 5.287878787878788e-06, "num_tokens": 2803437.0, "completions/mean_length": 87.625, "completions/min_length": 79.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.625, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.8798445463180542, "rewards/meter/std": 0.19600093364715576, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9570155143737793, "rewards/repeat_soft/std": 0.033770255744457245, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7676316499710083, "rewards/total_composite/std": 0.08693553507328033, "reward": 0.7676316499710083, "reward_std": 0.08693552762269974, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12963920831680298, "sampling/sampling_logp_difference/max": 3.380307674407959, "sampling/importance_sampling_ratio/min": 0.03403697907924652, "sampling/importance_sampling_ratio/mean": 1.0035719871520996, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6941117271780968, "clip_ratio/low_mean": 0.028457446955144405, "clip_ratio/low_min": 0.028457446955144405, "clip_ratio/high_mean": 0.08605522383004427, "clip_ratio/high_max": 0.08605522383004427, "clip_ratio/region_mean": 0.11451267078518867, "reward_total_mean": 0.7676316499710083, "reward_meter_mean": 0.8798445463180542, "reward_meter_std": 0.19600093364715576, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9570155143737793, "reward_repeat_soft_std": 0.033770255744457245, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7676316499710083, "reward_total_composite_std": 0.08693553507328033} {"timestamp_utc": "2026-04-13T02:13:09Z", "mode": "train", "global_step": 1557, "epoch": 0.15640381717729784, "loss": 0.0155, "grad_norm": 8.665514945983887, "learning_rate": 5.284848484848485e-06, "num_tokens": 2805713.0, "completions/mean_length": 112.5, "completions/min_length": 95.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.5, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.8759359121322632, "rewards/meter/std": 0.10172134637832642, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9541489481925964, "rewards/repeat_soft/std": 0.01881321147084236, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7528360486030579, "rewards/total_composite/std": 0.046818725764751434, "reward": 0.7528360486030579, "reward_std": 0.04681873321533203, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11819612234830856, "sampling/sampling_logp_difference/max": 1.5964171886444092, "sampling/importance_sampling_ratio/min": 0.2761373519897461, "sampling/importance_sampling_ratio/mean": 1.0185542106628418, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.813840351998806, "clip_ratio/low_mean": 0.05618071183562279, "clip_ratio/low_min": 0.05618071183562279, "clip_ratio/high_mean": 0.07006645482033491, "clip_ratio/high_max": 0.07006645482033491, "clip_ratio/region_mean": 0.1262471666559577, "reward_total_mean": 0.7528360486030579, "reward_meter_mean": 0.8759359121322632, "reward_meter_std": 0.10172134637832642, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9541489481925964, "reward_repeat_soft_std": 0.01881321147084236, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7528360486030579, "reward_total_composite_std": 0.046818725764751434} {"timestamp_utc": "2026-04-13T02:13:15Z", "mode": "train", "global_step": 1558, "epoch": 0.15650426921145152, "loss": 0.0471, "grad_norm": 19.224754333496094, "learning_rate": 5.281818181818183e-06, "num_tokens": 2807399.0, "completions/mean_length": 45.75, "completions/min_length": 41.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.75, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9902905225753784, "rewards/meter/std": 0.005174807272851467, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9818785786628723, "rewards/repeat_soft/std": 0.02425629459321499, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.12906257808208466, "rewards/total_composite/mean": 0.8258185982704163, "rewards/total_composite/std": 0.039238568395376205, "reward": 0.8258185982704163, "reward_std": 0.039238572120666504, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11450855433940887, "sampling/sampling_logp_difference/max": 1.6495556831359863, "sampling/importance_sampling_ratio/min": 0.1921352744102478, "sampling/importance_sampling_ratio/mean": 1.0055688619613647, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6227264404296875, "clip_ratio/low_mean": 0.07373395585455, "clip_ratio/low_min": 0.07373395585455, "clip_ratio/high_mean": 0.024950592778623104, "clip_ratio/high_max": 0.024950592778623104, "clip_ratio/region_mean": 0.09868454863317311, "reward_total_mean": 0.8258185982704163, "reward_meter_mean": 0.9902905225753784, "reward_meter_std": 0.005174807272851467, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9818785786628723, "reward_repeat_soft_std": 0.02425629459321499, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.12906257808208466, "reward_total_composite_mean": 0.8258185982704163, "reward_total_composite_std": 0.039238568395376205} {"timestamp_utc": "2026-04-13T02:13:22Z", "mode": "train", "global_step": 1559, "epoch": 0.15660472124560523, "loss": 0.0251, "grad_norm": 27.52552604675293, "learning_rate": 5.278787878787879e-06, "num_tokens": 2808933.0, "completions/mean_length": 32.75, "completions/min_length": 29.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.75, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9669034481048584, "rewards/meter/std": 0.05016957223415375, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9615821838378906, "rewards/repeat_soft/std": 0.002596014179289341, "rewards/judge_quality/mean": 0.44999998807907104, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8162647485733032, "rewards/total_composite/std": 0.02253422886133194, "reward": 0.8162647485733032, "reward_std": 0.022534240037202835, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15355679392814636, "sampling/sampling_logp_difference/max": 1.6701269149780273, "sampling/importance_sampling_ratio/min": 0.18822316825389862, "sampling/importance_sampling_ratio/mean": 1.022071123123169, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9319895654916763, "clip_ratio/low_mean": 0.026819923892617226, "clip_ratio/low_min": 0.026819923892617226, "clip_ratio/high_mean": 0.10322045534849167, "clip_ratio/high_max": 0.10322045534849167, "clip_ratio/region_mean": 0.1300403792411089, "reward_total_mean": 0.8162647485733032, "reward_meter_mean": 0.9669034481048584, "reward_meter_std": 0.05016957223415375, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9615821838378906, "reward_repeat_soft_std": 0.002596014179289341, "reward_judge_quality_mean": 0.44999998807907104, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8162647485733032, "reward_total_composite_std": 0.02253422886133194} {"timestamp_utc": "2026-04-13T02:13:29Z", "mode": "train", "global_step": 1560, "epoch": 0.1567051732797589, "loss": 0.0244, "grad_norm": 10.929461479187012, "learning_rate": 5.2757575757575764e-06, "num_tokens": 2810616.0, "completions/mean_length": 59.375, "completions/min_length": 51.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.375, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9711092114448547, "rewards/meter/std": 0.046040091663599014, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9710990786552429, "rewards/repeat_soft/std": 0.021918797865509987, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.8146090507507324, "rewards/total_composite/std": 0.023818375542759895, "reward": 0.8146090507507324, "reward_std": 0.0238183606415987, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12488669157028198, "sampling/sampling_logp_difference/max": 1.5103936195373535, "sampling/importance_sampling_ratio/min": 0.22082304954528809, "sampling/importance_sampling_ratio/mean": 1.0181398391723633, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7429208904504776, "clip_ratio/low_mean": 0.0374409556388855, "clip_ratio/low_min": 0.0374409556388855, "clip_ratio/high_mean": 0.0897889668121934, "clip_ratio/high_max": 0.0897889668121934, "clip_ratio/region_mean": 0.1272299224510789, "reward_total_mean": 0.8146090507507324, "reward_meter_mean": 0.9711092114448547, "reward_meter_std": 0.046040091663599014, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9710990786552429, "reward_repeat_soft_std": 0.021918797865509987, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.8146090507507324, "reward_total_composite_std": 0.023818375542759895} {"timestamp_utc": "2026-04-13T02:13:36Z", "mode": "train", "global_step": 1561, "epoch": 0.1568056253139126, "loss": 0.0334, "grad_norm": 16.403230667114258, "learning_rate": 5.272727272727273e-06, "num_tokens": 2812355.0, "completions/mean_length": 57.375, "completions/min_length": 49.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.375, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.8054253458976746, "rewards/meter/std": 0.3002900779247284, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9825705289840698, "rewards/repeat_soft/std": 0.019382895901799202, "rewards/judge_quality/mean": 0.7400000095367432, "rewards/judge_quality/std": 0.24859607219696045, "rewards/total_composite/mean": 0.8326984643936157, "rewards/total_composite/std": 0.19689732789993286, "reward": 0.8326984643936157, "reward_std": 0.19689732789993286, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1445525884628296, "sampling/sampling_logp_difference/max": 2.1695475578308105, "sampling/importance_sampling_ratio/min": 0.11422928422689438, "sampling/importance_sampling_ratio/mean": 0.9917795062065125, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8385416865348816, "clip_ratio/low_mean": 0.03646856639534235, "clip_ratio/low_min": 0.03646856639534235, "clip_ratio/high_mean": 0.08579504676163197, "clip_ratio/high_max": 0.08579504676163197, "clip_ratio/region_mean": 0.12226361315697432, "reward_total_mean": 0.8326984643936157, "reward_meter_mean": 0.8054253458976746, "reward_meter_std": 0.3002900779247284, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9825705289840698, "reward_repeat_soft_std": 0.019382895901799202, "reward_judge_quality_mean": 0.7400000095367432, "reward_judge_quality_std": 0.24859607219696045, "reward_total_composite_mean": 0.8326984643936157, "reward_total_composite_std": 0.19689732789993286} {"timestamp_utc": "2026-04-13T02:13:44Z", "mode": "train", "global_step": 1562, "epoch": 0.1569060773480663, "loss": 0.086, "grad_norm": 18.107629776000977, "learning_rate": 5.26969696969697e-06, "num_tokens": 2814467.0, "completions/mean_length": 73.0, "completions/min_length": 64.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.8456262350082397, "rewards/meter/std": 0.2694758176803589, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8965013027191162, "rewards/repeat_soft/std": 0.06247805804014206, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7473069429397583, "rewards/total_composite/std": 0.12090533971786499, "reward": 0.7473069429397583, "reward_std": 0.12090533971786499, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15124297142028809, "sampling/sampling_logp_difference/max": 2.7755603790283203, "sampling/importance_sampling_ratio/min": 0.06231454759836197, "sampling/importance_sampling_ratio/mean": 0.9849960207939148, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6036535277962685, "clip_ratio/low_mean": 0.033823530189692974, "clip_ratio/low_min": 0.033823530189692974, "clip_ratio/high_mean": 0.09825635235756636, "clip_ratio/high_max": 0.09825635235756636, "clip_ratio/region_mean": 0.13207988254725933, "reward_total_mean": 0.7473069429397583, "reward_meter_mean": 0.8456262350082397, "reward_meter_std": 0.2694758176803589, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8965013027191162, "reward_repeat_soft_std": 0.06247805804014206, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7473069429397583, "reward_total_composite_std": 0.12090533971786499} {"timestamp_utc": "2026-04-13T02:13:52Z", "mode": "train", "global_step": 1563, "epoch": 0.15700652938221998, "loss": 0.0366, "grad_norm": 7.651218414306641, "learning_rate": 5.2666666666666665e-06, "num_tokens": 2816882.0, "completions/mean_length": 118.875, "completions/min_length": 103.0, "completions/max_length": 144.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.875, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 144.0, "rewards/meter/mean": 0.975464940071106, "rewards/meter/std": 0.020047660917043686, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8819175958633423, "rewards/repeat_soft/std": 0.08132343739271164, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.8144009709358215, "rewards/total_composite/std": 0.037643831223249435, "reward": 0.8144009709358215, "reward_std": 0.03764383867383003, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11741794645786285, "sampling/sampling_logp_difference/max": 2.8427233695983887, "sampling/importance_sampling_ratio/min": 0.058266766369342804, "sampling/importance_sampling_ratio/mean": 1.0020719766616821, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6424183025956154, "clip_ratio/low_mean": 0.0823424351401627, "clip_ratio/low_min": 0.0823424351401627, "clip_ratio/high_mean": 0.027992278337478638, "clip_ratio/high_max": 0.027992278337478638, "clip_ratio/region_mean": 0.11033471347764134, "reward_total_mean": 0.8144009709358215, "reward_meter_mean": 0.975464940071106, "reward_meter_std": 0.020047660917043686, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8819175958633423, "reward_repeat_soft_std": 0.08132343739271164, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.8144009709358215, "reward_total_composite_std": 0.037643831223249435} {"timestamp_utc": "2026-04-13T02:13:58Z", "mode": "train", "global_step": 1564, "epoch": 0.1571069814163737, "loss": 0.0064, "grad_norm": 13.394765853881836, "learning_rate": 5.263636363636364e-06, "num_tokens": 2818355.0, "completions/mean_length": 36.125, "completions/min_length": 32.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.125, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.995460033416748, "rewards/meter/std": 0.0048712920397520065, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9560462832450867, "rewards/repeat_soft/std": 0.011949889361858368, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8206866383552551, "rewards/total_composite/std": 0.004456805065274239, "reward": 0.8206866383552551, "reward_std": 0.004456816241145134, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12295495718717575, "sampling/sampling_logp_difference/max": 1.3528213500976562, "sampling/importance_sampling_ratio/min": 0.25850987434387207, "sampling/importance_sampling_ratio/mean": 1.0089950561523438, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8114751502871513, "clip_ratio/low_mean": 0.08069086447358131, "clip_ratio/low_min": 0.08069086447358131, "clip_ratio/high_mean": 0.04361201962456107, "clip_ratio/high_max": 0.04361201962456107, "clip_ratio/region_mean": 0.12430288409814239, "reward_total_mean": 0.8206866383552551, "reward_meter_mean": 0.995460033416748, "reward_meter_std": 0.0048712920397520065, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9560462832450867, "reward_repeat_soft_std": 0.011949889361858368, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8206866383552551, "reward_total_composite_std": 0.004456805065274239} {"timestamp_utc": "2026-04-13T02:14:05Z", "mode": "train", "global_step": 1565, "epoch": 0.15720743345052737, "loss": 0.0418, "grad_norm": 13.21590518951416, "learning_rate": 5.26060606060606e-06, "num_tokens": 2819925.0, "completions/mean_length": 46.25, "completions/min_length": 42.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.25, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.6447970867156982, "rewards/meter/std": 0.3395019769668579, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9918428659439087, "rewards/repeat_soft/std": 0.009806911461055279, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.6874679327011108, "rewards/total_composite/std": 0.18237470090389252, "reward": 0.6874679327011108, "reward_std": 0.18237468600273132, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13570694625377655, "sampling/sampling_logp_difference/max": 2.6441245079040527, "sampling/importance_sampling_ratio/min": 0.07106754928827286, "sampling/importance_sampling_ratio/mean": 0.9828737378120422, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6426043398678303, "clip_ratio/low_mean": 0.053833167999982834, "clip_ratio/low_min": 0.053833167999982834, "clip_ratio/high_mean": 0.07970016542822123, "clip_ratio/high_max": 0.07970016542822123, "clip_ratio/region_mean": 0.13353333342820406, "reward_total_mean": 0.6874679327011108, "reward_meter_mean": 0.6447970867156982, "reward_meter_std": 0.3395019769668579, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9918428659439087, "reward_repeat_soft_std": 0.009806911461055279, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.6874679327011108, "reward_total_composite_std": 0.18237470090389252} {"timestamp_utc": "2026-04-13T02:14:24Z", "mode": "train", "global_step": 1566, "epoch": 0.15730788548468105, "loss": -0.0774, "grad_norm": 3.955120801925659, "learning_rate": 5.257575757575758e-06, "num_tokens": 2821411.0, "completions/mean_length": 96.75, "completions/min_length": 33.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 37.42857360839844, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.38198208808898926, "rewards/meter/std": 0.30076706409454346, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9843258261680603, "rewards/repeat_soft/std": 0.022214235737919807, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.23439893126487732, "rewards/total_composite/mean": 0.49214422702789307, "rewards/total_composite/std": 0.2606990933418274, "reward": 0.49214422702789307, "reward_std": 0.260699063539505, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1382083147764206, "sampling/sampling_logp_difference/max": 1.7980031967163086, "sampling/importance_sampling_ratio/min": 0.1656292974948883, "sampling/importance_sampling_ratio/mean": 1.0129616260528564, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6160364001989365, "clip_ratio/low_mean": 0.0779057014733553, "clip_ratio/low_min": 0.0779057014733553, "clip_ratio/high_mean": 0.08485020138323307, "clip_ratio/high_max": 0.08485020138323307, "clip_ratio/region_mean": 0.16275590285658836, "reward_total_mean": 0.49214422702789307, "reward_meter_mean": 0.38198208808898926, "reward_meter_std": 0.30076706409454346, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9843258261680603, "reward_repeat_soft_std": 0.022214235737919807, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.23439893126487732, "reward_total_composite_mean": 0.49214422702789307, "reward_total_composite_std": 0.2606990933418274} {"timestamp_utc": "2026-04-13T02:14:31Z", "mode": "train", "global_step": 1567, "epoch": 0.15740833751883476, "loss": 0.0042, "grad_norm": 13.620513916015625, "learning_rate": 5.2545454545454555e-06, "num_tokens": 2823050.0, "completions/mean_length": 52.875, "completions/min_length": 45.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.875, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.5991332530975342, "rewards/meter/std": 0.44140660762786865, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9806076288223267, "rewards/repeat_soft/std": 0.030033603310585022, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.563369870185852, "rewards/total_composite/std": 0.31181997060775757, "reward": 0.563369870185852, "reward_std": 0.31181997060775757, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16100645065307617, "sampling/sampling_logp_difference/max": 1.7384967803955078, "sampling/importance_sampling_ratio/min": 0.17578443884849548, "sampling/importance_sampling_ratio/mean": 0.9931136965751648, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0531306639313698, "clip_ratio/low_mean": 0.0798157025128603, "clip_ratio/low_min": 0.0798157025128603, "clip_ratio/high_mean": 0.07412173319607973, "clip_ratio/high_max": 0.07412173319607973, "clip_ratio/region_mean": 0.15393743570894003, "reward_total_mean": 0.563369870185852, "reward_meter_mean": 0.5991332530975342, "reward_meter_std": 0.44140660762786865, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9806076288223267, "reward_repeat_soft_std": 0.030033603310585022, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.563369870185852, "reward_total_composite_std": 0.31181997060775757} {"timestamp_utc": "2026-04-13T02:14:38Z", "mode": "train", "global_step": 1568, "epoch": 0.15750878955298844, "loss": 0.0678, "grad_norm": 13.213634490966797, "learning_rate": 5.251515151515152e-06, "num_tokens": 2824579.0, "completions/mean_length": 46.125, "completions/min_length": 39.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.125, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.7119094133377075, "rewards/meter/std": 0.35641399025917053, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9364505410194397, "rewards/repeat_soft/std": 0.05971188843250275, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.6911292672157288, "rewards/total_composite/std": 0.15544041991233826, "reward": 0.6911292672157288, "reward_std": 0.15544040501117706, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10100485384464264, "sampling/sampling_logp_difference/max": 1.3811063766479492, "sampling/importance_sampling_ratio/min": 0.2513003945350647, "sampling/importance_sampling_ratio/mean": 1.0055468082427979, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.574658241122961, "clip_ratio/low_mean": 0.04077593516558409, "clip_ratio/low_min": 0.04077593516558409, "clip_ratio/high_mean": 0.03337606834247708, "clip_ratio/high_max": 0.03337606834247708, "clip_ratio/region_mean": 0.07415200350806117, "reward_total_mean": 0.6911292672157288, "reward_meter_mean": 0.7119094133377075, "reward_meter_std": 0.35641399025917053, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9364505410194397, "reward_repeat_soft_std": 0.05971188843250275, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.6911292672157288, "reward_total_composite_std": 0.15544041991233826} {"timestamp_utc": "2026-04-13T02:14:45Z", "mode": "train", "global_step": 1569, "epoch": 0.15760924158714215, "loss": -0.001, "grad_norm": 14.432804107666016, "learning_rate": 5.248484848484849e-06, "num_tokens": 2826250.0, "completions/mean_length": 35.875, "completions/min_length": 25.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.875, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.8289706707000732, "rewards/meter/std": 0.2509479224681854, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9434381723403931, "rewards/repeat_soft/std": 0.048494815826416016, "rewards/judge_quality/mean": 0.6312500238418579, "rewards/judge_quality/std": 0.22730958461761475, "rewards/total_composite/mean": 0.8067556023597717, "rewards/total_composite/std": 0.15220704674720764, "reward": 0.8067556023597717, "reward_std": 0.15220704674720764, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14686079323291779, "sampling/sampling_logp_difference/max": 2.328439712524414, "sampling/importance_sampling_ratio/min": 0.09744767099618912, "sampling/importance_sampling_ratio/mean": 0.9909419417381287, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5519997514784336, "clip_ratio/low_mean": 0.06788461655378342, "clip_ratio/low_min": 0.06788461655378342, "clip_ratio/high_mean": 0.06899840570986271, "clip_ratio/high_max": 0.06899840570986271, "clip_ratio/region_mean": 0.13688302226364613, "reward_total_mean": 0.8067556023597717, "reward_meter_mean": 0.8289706707000732, "reward_meter_std": 0.2509479224681854, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9434381723403931, "reward_repeat_soft_std": 0.048494815826416016, "reward_judge_quality_mean": 0.6312500238418579, "reward_judge_quality_std": 0.22730958461761475, "reward_total_composite_mean": 0.8067556023597717, "reward_total_composite_std": 0.15220704674720764} {"timestamp_utc": "2026-04-13T02:14:53Z", "mode": "train", "global_step": 1570, "epoch": 0.15770969362129583, "loss": 0.023, "grad_norm": 6.778067111968994, "learning_rate": 5.245454545454546e-06, "num_tokens": 2828929.0, "completions/mean_length": 146.875, "completions/min_length": 112.0, "completions/max_length": 166.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 146.875, "completions/min_terminated_length": 112.0, "completions/max_terminated_length": 166.0, "rewards/meter/mean": 0.9836880564689636, "rewards/meter/std": 0.01420046016573906, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7380791306495667, "rewards/repeat_soft/std": 0.04685848206281662, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.760092556476593, "rewards/total_composite/std": 0.03317518159747124, "reward": 0.760092556476593, "reward_std": 0.03317520394921303, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09881823509931564, "sampling/sampling_logp_difference/max": 3.2496654987335205, "sampling/importance_sampling_ratio/min": 0.03878718242049217, "sampling/importance_sampling_ratio/mean": 1.0019347667694092, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5475426316261292, "clip_ratio/low_mean": 0.03796008788049221, "clip_ratio/low_min": 0.03796008788049221, "clip_ratio/high_mean": 0.05367311090230942, "clip_ratio/high_max": 0.05367311090230942, "clip_ratio/region_mean": 0.09163319878280163, "reward_total_mean": 0.760092556476593, "reward_meter_mean": 0.9836880564689636, "reward_meter_std": 0.01420046016573906, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7380791306495667, "reward_repeat_soft_std": 0.04685848206281662, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.760092556476593, "reward_total_composite_std": 0.03317518159747124} {"timestamp_utc": "2026-04-13T02:14:59Z", "mode": "train", "global_step": 1571, "epoch": 0.1578101456554495, "loss": 0.02, "grad_norm": 19.600601196289062, "learning_rate": 5.242424242424244e-06, "num_tokens": 2830592.0, "completions/mean_length": 45.875, "completions/min_length": 39.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.875, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.9536913633346558, "rewards/meter/std": 0.04488219693303108, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9234068393707275, "rewards/repeat_soft/std": 0.04520651698112488, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.800876796245575, "rewards/total_composite/std": 0.020510884001851082, "reward": 0.800876796245575, "reward_std": 0.02051088958978653, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1429482400417328, "sampling/sampling_logp_difference/max": 3.0157933235168457, "sampling/importance_sampling_ratio/min": 0.04900693893432617, "sampling/importance_sampling_ratio/mean": 1.0281789302825928, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7933443263173103, "clip_ratio/low_mean": 0.04815217386931181, "clip_ratio/low_min": 0.04815217386931181, "clip_ratio/high_mean": 0.07709801197052002, "clip_ratio/high_max": 0.07709801197052002, "clip_ratio/region_mean": 0.12525018583983183, "reward_total_mean": 0.800876796245575, "reward_meter_mean": 0.9536913633346558, "reward_meter_std": 0.04488219693303108, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9234068393707275, "reward_repeat_soft_std": 0.04520651698112488, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.800876796245575, "reward_total_composite_std": 0.020510884001851082} {"timestamp_utc": "2026-04-13T02:15:06Z", "mode": "train", "global_step": 1572, "epoch": 0.15791059768960322, "loss": 0.0337, "grad_norm": 18.425466537475586, "learning_rate": 5.23939393939394e-06, "num_tokens": 2832174.0, "completions/mean_length": 39.75, "completions/min_length": 36.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.75, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.6533671617507935, "rewards/meter/std": 0.41411831974983215, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9899790287017822, "rewards/repeat_soft/std": 0.024572059512138367, "rewards/judge_quality/mean": 0.6025000214576721, "rewards/judge_quality/std": 0.19099366664886475, "rewards/total_composite/mean": 0.723763108253479, "rewards/total_composite/std": 0.19085782766342163, "reward": 0.723763108253479, "reward_std": 0.19085781276226044, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12411440163850784, "sampling/sampling_logp_difference/max": 2.7849111557006836, "sampling/importance_sampling_ratio/min": 0.06173457205295563, "sampling/importance_sampling_ratio/mean": 1.0167231559753418, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5410930141806602, "clip_ratio/low_mean": 0.04430417250841856, "clip_ratio/low_min": 0.04430417250841856, "clip_ratio/high_mean": 0.06363866524770856, "clip_ratio/high_max": 0.06363866524770856, "clip_ratio/region_mean": 0.10794283775612712, "reward_total_mean": 0.723763108253479, "reward_meter_mean": 0.6533671617507935, "reward_meter_std": 0.41411831974983215, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9899790287017822, "reward_repeat_soft_std": 0.024572059512138367, "reward_judge_quality_mean": 0.6025000214576721, "reward_judge_quality_std": 0.19099366664886475, "reward_total_composite_mean": 0.723763108253479, "reward_total_composite_std": 0.19085782766342163} {"timestamp_utc": "2026-04-13T02:15:13Z", "mode": "train", "global_step": 1573, "epoch": 0.1580110497237569, "loss": 0.0244, "grad_norm": 13.309405326843262, "learning_rate": 5.236363636363637e-06, "num_tokens": 2833855.0, "completions/mean_length": 60.125, "completions/min_length": 50.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.125, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.8562027812004089, "rewards/meter/std": 0.28945204615592957, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.987561821937561, "rewards/repeat_soft/std": 0.010134817101061344, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.797547459602356, "rewards/total_composite/std": 0.15376697480678558, "reward": 0.797547459602356, "reward_std": 0.1537669599056244, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17131894826889038, "sampling/sampling_logp_difference/max": 1.739485740661621, "sampling/importance_sampling_ratio/min": 0.17561069130897522, "sampling/importance_sampling_ratio/mean": 1.0227282047271729, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1721694096922874, "clip_ratio/low_mean": 0.02291666716337204, "clip_ratio/low_min": 0.02291666716337204, "clip_ratio/high_mean": 0.15081149898469448, "clip_ratio/high_max": 0.15081149898469448, "clip_ratio/region_mean": 0.17372816614806652, "reward_total_mean": 0.797547459602356, "reward_meter_mean": 0.8562027812004089, "reward_meter_std": 0.28945204615592957, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.987561821937561, "reward_repeat_soft_std": 0.010134817101061344, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.797547459602356, "reward_total_composite_std": 0.15376697480678558} {"timestamp_utc": "2026-04-13T02:15:23Z", "mode": "train", "global_step": 1574, "epoch": 0.1581115017579106, "loss": 0.0479, "grad_norm": 9.460936546325684, "learning_rate": 5.233333333333334e-06, "num_tokens": 2835918.0, "completions/mean_length": 96.875, "completions/min_length": 86.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.875, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.9805595278739929, "rewards/meter/std": 0.016498932614922523, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9665409326553345, "rewards/repeat_soft/std": 0.02343418076634407, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8139058351516724, "rewards/total_composite/std": 0.0075031546875834465, "reward": 0.8139058351516724, "reward_std": 0.007503171917051077, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13241924345493317, "sampling/sampling_logp_difference/max": 1.5342906713485718, "sampling/importance_sampling_ratio/min": 0.21560858190059662, "sampling/importance_sampling_ratio/mean": 1.0250310897827148, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9004007950425148, "clip_ratio/low_mean": 0.034539804328233004, "clip_ratio/low_min": 0.034539804328233004, "clip_ratio/high_mean": 0.057341497391462326, "clip_ratio/high_max": 0.057341497391462326, "clip_ratio/region_mean": 0.09188130171969533, "reward_total_mean": 0.8139058351516724, "reward_meter_mean": 0.9805595278739929, "reward_meter_std": 0.016498932614922523, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9665409326553345, "reward_repeat_soft_std": 0.02343418076634407, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8139058351516724, "reward_total_composite_std": 0.0075031546875834465} {"timestamp_utc": "2026-04-13T02:15:35Z", "mode": "train", "global_step": 1575, "epoch": 0.1582119537920643, "loss": -0.1075, "grad_norm": 2.6981868743896484, "learning_rate": 5.230303030303031e-06, "num_tokens": 2837460.0, "completions/mean_length": 106.75, "completions/min_length": 41.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 48.85714340209961, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.7494562268257141, "rewards/meter/std": 0.3602370321750641, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9859874248504639, "rewards/repeat_soft/std": 0.012168104760348797, "rewards/judge_quality/mean": 0.4925000071525574, "rewards/judge_quality/std": 0.30103394389152527, "rewards/total_composite/mean": 0.7028156518936157, "rewards/total_composite/std": 0.3123655319213867, "reward": 0.7028156518936157, "reward_std": 0.3123655319213867, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1561298668384552, "sampling/sampling_logp_difference/max": 2.37917423248291, "sampling/importance_sampling_ratio/min": 0.09262703359127045, "sampling/importance_sampling_ratio/mean": 0.9775755405426025, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4697025455534458, "clip_ratio/low_mean": 0.023706896230578423, "clip_ratio/low_min": 0.023706896230578423, "clip_ratio/high_mean": 0.0963577888906002, "clip_ratio/high_max": 0.0963577888906002, "clip_ratio/region_mean": 0.12006468512117863, "reward_total_mean": 0.7028156518936157, "reward_meter_mean": 0.7494562268257141, "reward_meter_std": 0.3602370321750641, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9859874248504639, "reward_repeat_soft_std": 0.012168104760348797, "reward_judge_quality_mean": 0.4925000071525574, "reward_judge_quality_std": 0.30103394389152527, "reward_total_composite_mean": 0.7028156518936157, "reward_total_composite_std": 0.3123655319213867} {"timestamp_utc": "2026-04-13T02:15:42Z", "mode": "train", "global_step": 1576, "epoch": 0.15831240582621797, "loss": 0.0375, "grad_norm": 8.541095733642578, "learning_rate": 5.2272727272727274e-06, "num_tokens": 2839371.0, "completions/mean_length": 77.875, "completions/min_length": 69.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.875, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.6606801152229309, "rewards/meter/std": 0.3551565706729889, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9631742238998413, "rewards/repeat_soft/std": 0.008145919069647789, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.674498438835144, "rewards/total_composite/std": 0.1723601222038269, "reward": 0.674498438835144, "reward_std": 0.1723601222038269, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1294480264186859, "sampling/sampling_logp_difference/max": 1.669367790222168, "sampling/importance_sampling_ratio/min": 0.18836610019207, "sampling/importance_sampling_ratio/mean": 1.0304369926452637, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9465383738279343, "clip_ratio/low_mean": 0.049793957732617855, "clip_ratio/low_min": 0.049793957732617855, "clip_ratio/high_mean": 0.09431310929358006, "clip_ratio/high_max": 0.09431310929358006, "clip_ratio/region_mean": 0.1441070670261979, "reward_total_mean": 0.674498438835144, "reward_meter_mean": 0.6606801152229309, "reward_meter_std": 0.3551565706729889, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9631742238998413, "reward_repeat_soft_std": 0.008145919069647789, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.674498438835144, "reward_total_composite_std": 0.1723601222038269} {"timestamp_utc": "2026-04-13T02:15:48Z", "mode": "train", "global_step": 1577, "epoch": 0.15841285786037168, "loss": 0.1191, "grad_norm": 12.666656494140625, "learning_rate": 5.224242424242425e-06, "num_tokens": 2840806.0, "completions/mean_length": 32.375, "completions/min_length": 27.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.375, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.5178497433662415, "rewards/meter/std": 0.3620348572731018, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9563978910446167, "rewards/repeat_soft/std": 0.03733189404010773, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.6080471277236938, "rewards/total_composite/std": 0.16184964776039124, "reward": 0.6080471277236938, "reward_std": 0.16184964776039124, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16670161485671997, "sampling/sampling_logp_difference/max": 2.906221628189087, "sampling/importance_sampling_ratio/min": 0.05468194931745529, "sampling/importance_sampling_ratio/mean": 1.0018057823181152, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5760586857795715, "clip_ratio/low_mean": 0.09763353783637285, "clip_ratio/low_min": 0.09763353783637285, "clip_ratio/high_mean": 0.06762566231191158, "clip_ratio/high_max": 0.06762566231191158, "clip_ratio/region_mean": 0.16525920014828444, "reward_total_mean": 0.6080471277236938, "reward_meter_mean": 0.5178497433662415, "reward_meter_std": 0.3620348572731018, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9563978910446167, "reward_repeat_soft_std": 0.03733189404010773, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.6080471277236938, "reward_total_composite_std": 0.16184964776039124} {"timestamp_utc": "2026-04-13T02:15:55Z", "mode": "train", "global_step": 1578, "epoch": 0.15851330989452536, "loss": -0.0688, "grad_norm": 14.825627326965332, "learning_rate": 5.221212121212121e-06, "num_tokens": 2842428.0, "completions/mean_length": 29.75, "completions/min_length": 24.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.75, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9861258864402771, "rewards/meter/std": 0.02343398705124855, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.959877610206604, "rewards/repeat_soft/std": 0.007417192216962576, "rewards/judge_quality/mean": 0.8362500667572021, "rewards/judge_quality/std": 0.23688077926635742, "rewards/total_composite/mean": 0.9406194090843201, "rewards/total_composite/std": 0.07187168300151825, "reward": 0.9406194090843201, "reward_std": 0.07187168300151825, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.134688600897789, "sampling/sampling_logp_difference/max": 2.0509486198425293, "sampling/importance_sampling_ratio/min": 0.12861284613609314, "sampling/importance_sampling_ratio/mean": 0.998718798160553, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9191323816776276, "clip_ratio/low_mean": 0.029761905781924725, "clip_ratio/low_min": 0.029761905781924725, "clip_ratio/high_mean": 0.10084845311939716, "clip_ratio/high_max": 0.10084845311939716, "clip_ratio/region_mean": 0.1306103589013219, "reward_total_mean": 0.9406194090843201, "reward_meter_mean": 0.9861258864402771, "reward_meter_std": 0.02343398705124855, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.959877610206604, "reward_repeat_soft_std": 0.007417192216962576, "reward_judge_quality_mean": 0.8362500667572021, "reward_judge_quality_std": 0.23688077926635742, "reward_total_composite_mean": 0.9406194090843201, "reward_total_composite_std": 0.07187168300151825} {"timestamp_utc": "2026-04-13T02:16:01Z", "mode": "train", "global_step": 1579, "epoch": 0.15861376192867904, "loss": 0.0377, "grad_norm": 11.670080184936523, "learning_rate": 5.218181818181819e-06, "num_tokens": 2844286.0, "completions/mean_length": 57.25, "completions/min_length": 52.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.25, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.24800460040569305, "rewards/meter/std": 0.19551372528076172, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9822443127632141, "rewards/repeat_soft/std": 0.01011641789227724, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4858265221118927, "rewards/total_composite/std": 0.08781725913286209, "reward": 0.4858265221118927, "reward_std": 0.08781726658344269, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1630091667175293, "sampling/sampling_logp_difference/max": 3.3870816230773926, "sampling/importance_sampling_ratio/min": 0.03380719572305679, "sampling/importance_sampling_ratio/mean": 1.0212352275848389, "sampling/importance_sampling_ratio/max": 1.893796682357788, "entropy": 0.9674319326877594, "clip_ratio/low_mean": 0.11094911023974419, "clip_ratio/low_min": 0.11094911023974419, "clip_ratio/high_mean": 0.04178459197282791, "clip_ratio/high_max": 0.04178459197282791, "clip_ratio/region_mean": 0.1527337022125721, "reward_total_mean": 0.4858265221118927, "reward_meter_mean": 0.24800460040569305, "reward_meter_std": 0.19551372528076172, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9822443127632141, "reward_repeat_soft_std": 0.01011641789227724, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4858265221118927, "reward_total_composite_std": 0.08781725913286209} {"timestamp_utc": "2026-04-13T02:16:09Z", "mode": "train", "global_step": 1580, "epoch": 0.15871421396283275, "loss": 0.0279, "grad_norm": 7.111854553222656, "learning_rate": 5.215151515151516e-06, "num_tokens": 2846626.0, "completions/mean_length": 125.5, "completions/min_length": 105.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.5, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.9945318102836609, "rewards/meter/std": 0.0038908564019948244, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9463728666305542, "rewards/repeat_soft/std": 0.04501591995358467, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.19949938356876373, "rewards/total_composite/mean": 0.8241766691207886, "rewards/total_composite/std": 0.06123065948486328, "reward": 0.8241766691207886, "reward_std": 0.06123065575957298, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12904635071754456, "sampling/sampling_logp_difference/max": 2.412154197692871, "sampling/importance_sampling_ratio/min": 0.08962202072143555, "sampling/importance_sampling_ratio/mean": 1.0061808824539185, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8689513653516769, "clip_ratio/low_mean": 0.0923598064109683, "clip_ratio/low_min": 0.0923598064109683, "clip_ratio/high_mean": 0.03993599861860275, "clip_ratio/high_max": 0.03993599861860275, "clip_ratio/region_mean": 0.13229580502957106, "reward_total_mean": 0.8241766691207886, "reward_meter_mean": 0.9945318102836609, "reward_meter_std": 0.0038908564019948244, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9463728666305542, "reward_repeat_soft_std": 0.04501591995358467, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.19949938356876373, "reward_total_composite_mean": 0.8241766691207886, "reward_total_composite_std": 0.06123065948486328} {"timestamp_utc": "2026-04-13T02:16:17Z", "mode": "train", "global_step": 1581, "epoch": 0.15881466599698643, "loss": 0.0605, "grad_norm": 11.185583114624023, "learning_rate": 5.212121212121213e-06, "num_tokens": 2848453.0, "completions/mean_length": 65.375, "completions/min_length": 54.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.375, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9910717010498047, "rewards/meter/std": 0.005243103485554457, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9778027534484863, "rewards/repeat_soft/std": 0.024434076622128487, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.8242624998092651, "rewards/total_composite/std": 0.0035883288364857435, "reward": 0.8242624998092651, "reward_std": 0.0035883227828890085, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1388266533613205, "sampling/sampling_logp_difference/max": 1.970740795135498, "sampling/importance_sampling_ratio/min": 0.13935358822345734, "sampling/importance_sampling_ratio/mean": 1.0263645648956299, "sampling/importance_sampling_ratio/max": 1.8407282829284668, "entropy": 0.9901214390993118, "clip_ratio/low_mean": 0.09162651374936104, "clip_ratio/low_min": 0.09162651374936104, "clip_ratio/high_mean": 0.04133913153782487, "clip_ratio/high_max": 0.04133913153782487, "clip_ratio/region_mean": 0.1329656452871859, "reward_total_mean": 0.8242624998092651, "reward_meter_mean": 0.9910717010498047, "reward_meter_std": 0.005243103485554457, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9778027534484863, "reward_repeat_soft_std": 0.024434076622128487, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.8242624998092651, "reward_total_composite_std": 0.0035883288364857435} {"timestamp_utc": "2026-04-13T02:16:24Z", "mode": "train", "global_step": 1582, "epoch": 0.15891511803114014, "loss": 0.0566, "grad_norm": 13.333572387695312, "learning_rate": 5.209090909090909e-06, "num_tokens": 2850171.0, "completions/mean_length": 53.75, "completions/min_length": 51.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.75, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.6375963687896729, "rewards/meter/std": 0.3195393979549408, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9605250358581543, "rewards/repeat_soft/std": 0.08833162486553192, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.692720890045166, "rewards/total_composite/std": 0.16347229480743408, "reward": 0.692720890045166, "reward_std": 0.16347229480743408, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13126643002033234, "sampling/sampling_logp_difference/max": 1.9980952739715576, "sampling/importance_sampling_ratio/min": 0.13559330999851227, "sampling/importance_sampling_ratio/mean": 0.9896647334098816, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6479920521378517, "clip_ratio/low_mean": 0.029727094806730747, "clip_ratio/low_min": 0.029727094806730747, "clip_ratio/high_mean": 0.08224505465477705, "clip_ratio/high_max": 0.08224505465477705, "clip_ratio/region_mean": 0.1119721494615078, "reward_total_mean": 0.692720890045166, "reward_meter_mean": 0.6375963687896729, "reward_meter_std": 0.3195393979549408, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9605250358581543, "reward_repeat_soft_std": 0.08833162486553192, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.692720890045166, "reward_total_composite_std": 0.16347229480743408} {"timestamp_utc": "2026-04-13T02:16:35Z", "mode": "train", "global_step": 1583, "epoch": 0.15901557006529382, "loss": -0.0937, "grad_norm": 3.44030499458313, "learning_rate": 5.2060606060606065e-06, "num_tokens": 2851679.0, "completions/mean_length": 97.5, "completions/min_length": 28.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 38.28571701049805, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.6114853024482727, "rewards/meter/std": 0.42592066526412964, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9405215978622437, "rewards/repeat_soft/std": 0.08920525014400482, "rewards/judge_quality/mean": 0.6024999618530273, "rewards/judge_quality/std": 0.31878116726875305, "rewards/total_composite/mean": 0.6672422289848328, "rewards/total_composite/std": 0.2991509437561035, "reward": 0.6672422289848328, "reward_std": 0.2991509735584259, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15535138547420502, "sampling/sampling_logp_difference/max": 1.8551664352416992, "sampling/importance_sampling_ratio/min": 0.15642690658569336, "sampling/importance_sampling_ratio/mean": 0.9996631741523743, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6315374225378036, "clip_ratio/low_mean": 0.03002149797976017, "clip_ratio/low_min": 0.03002149797976017, "clip_ratio/high_mean": 0.0860129757784307, "clip_ratio/high_max": 0.0860129757784307, "clip_ratio/region_mean": 0.11603447375819087, "reward_total_mean": 0.6672422289848328, "reward_meter_mean": 0.6114853024482727, "reward_meter_std": 0.42592066526412964, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9405215978622437, "reward_repeat_soft_std": 0.08920525014400482, "reward_judge_quality_mean": 0.6024999618530273, "reward_judge_quality_std": 0.31878116726875305, "reward_total_composite_mean": 0.6672422289848328, "reward_total_composite_std": 0.2991509437561035} {"timestamp_utc": "2026-04-13T02:16:47Z", "mode": "train", "global_step": 1584, "epoch": 0.1591160220994475, "loss": -0.1614, "grad_norm": 2.6890056133270264, "learning_rate": 5.203030303030303e-06, "num_tokens": 2853522.0, "completions/mean_length": 135.375, "completions/min_length": 74.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 81.5714340209961, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.654733419418335, "rewards/meter/std": 0.3080720901489258, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9655880928039551, "rewards/repeat_soft/std": 0.03373369947075844, "rewards/judge_quality/mean": 0.4112499952316284, "rewards/judge_quality/std": 0.1797965168952942, "rewards/total_composite/mean": 0.6228674650192261, "rewards/total_composite/std": 0.2670319676399231, "reward": 0.6228674650192261, "reward_std": 0.2670319676399231, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11097462475299835, "sampling/sampling_logp_difference/max": 1.980720043182373, "sampling/importance_sampling_ratio/min": 0.2060966044664383, "sampling/importance_sampling_ratio/mean": 1.0104244947433472, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4421101547777653, "clip_ratio/low_mean": 0.02484917175024748, "clip_ratio/low_min": 0.02484917175024748, "clip_ratio/high_mean": 0.07286144234240055, "clip_ratio/high_max": 0.07286144234240055, "clip_ratio/region_mean": 0.09771061409264803, "reward_total_mean": 0.6228674650192261, "reward_meter_mean": 0.654733419418335, "reward_meter_std": 0.3080720901489258, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9655880928039551, "reward_repeat_soft_std": 0.03373369947075844, "reward_judge_quality_mean": 0.4112499952316284, "reward_judge_quality_std": 0.1797965168952942, "reward_total_composite_mean": 0.6228674650192261, "reward_total_composite_std": 0.2670319676399231} {"timestamp_utc": "2026-04-13T02:16:55Z", "mode": "train", "global_step": 1585, "epoch": 0.1592164741336012, "loss": 0.0298, "grad_norm": 10.402949333190918, "learning_rate": 5.2e-06, "num_tokens": 2855453.0, "completions/mean_length": 64.375, "completions/min_length": 52.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.375, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.899320125579834, "rewards/meter/std": 0.25145232677459717, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9406818151473999, "rewards/repeat_soft/std": 0.06331042945384979, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.8047622442245483, "rewards/total_composite/std": 0.1366616189479828, "reward": 0.8047622442245483, "reward_std": 0.1366616189479828, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13389858603477478, "sampling/sampling_logp_difference/max": 1.3220224380493164, "sampling/importance_sampling_ratio/min": 0.2665955722332001, "sampling/importance_sampling_ratio/mean": 1.0143699645996094, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0868324935436249, "clip_ratio/low_mean": 0.011538461782038212, "clip_ratio/low_min": 0.011538461782038212, "clip_ratio/high_mean": 0.09393923031166196, "clip_ratio/high_max": 0.09393923031166196, "clip_ratio/region_mean": 0.10547769209370017, "reward_total_mean": 0.8047622442245483, "reward_meter_mean": 0.899320125579834, "reward_meter_std": 0.25145232677459717, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9406818151473999, "reward_repeat_soft_std": 0.06331042945384979, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.8047622442245483, "reward_total_composite_std": 0.1366616189479828} {"timestamp_utc": "2026-04-13T02:17:01Z", "mode": "train", "global_step": 1586, "epoch": 0.1593169261677549, "loss": 0.1077, "grad_norm": 20.912981033325195, "learning_rate": 5.196969696969697e-06, "num_tokens": 2856906.0, "completions/mean_length": 38.625, "completions/min_length": 34.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.625, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.824360728263855, "rewards/meter/std": 0.2829618752002716, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9333338737487793, "rewards/repeat_soft/std": 0.049037300050258636, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.8340457081794739, "rewards/total_composite/std": 0.17116917669773102, "reward": 0.8340457081794739, "reward_std": 0.17116916179656982, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09579651057720184, "sampling/sampling_logp_difference/max": 1.4665424823760986, "sampling/importance_sampling_ratio/min": 0.2307218313217163, "sampling/importance_sampling_ratio/mean": 1.0153571367263794, "sampling/importance_sampling_ratio/max": 1.9631671905517578, "entropy": 0.4878162778913975, "clip_ratio/low_mean": 0.03622538037598133, "clip_ratio/low_min": 0.03622538037598133, "clip_ratio/high_mean": 0.04573412844911218, "clip_ratio/high_max": 0.04573412844911218, "clip_ratio/region_mean": 0.08195950882509351, "reward_total_mean": 0.8340457081794739, "reward_meter_mean": 0.824360728263855, "reward_meter_std": 0.2829618752002716, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9333338737487793, "reward_repeat_soft_std": 0.049037300050258636, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.8340457081794739, "reward_total_composite_std": 0.17116917669773102} {"timestamp_utc": "2026-04-13T02:17:08Z", "mode": "train", "global_step": 1587, "epoch": 0.1594173782019086, "loss": 0.0839, "grad_norm": 17.460983276367188, "learning_rate": 5.193939393939395e-06, "num_tokens": 2858419.0, "completions/mean_length": 38.125, "completions/min_length": 32.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.125, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.6031774878501892, "rewards/meter/std": 0.3062264323234558, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9900212287902832, "rewards/repeat_soft/std": 0.011030333116650581, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.2499571591615677, "rewards/total_composite/mean": 0.6786819696426392, "rewards/total_composite/std": 0.13810473680496216, "reward": 0.6786819696426392, "reward_std": 0.13810475170612335, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18406735360622406, "sampling/sampling_logp_difference/max": 2.5264105796813965, "sampling/importance_sampling_ratio/min": 0.07994545996189117, "sampling/importance_sampling_ratio/mean": 0.9867894053459167, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8641835898160934, "clip_ratio/low_mean": 0.04537927359342575, "clip_ratio/low_min": 0.04537927359342575, "clip_ratio/high_mean": 0.11394537333399057, "clip_ratio/high_max": 0.11394537333399057, "clip_ratio/region_mean": 0.15932464692741632, "reward_total_mean": 0.6786819696426392, "reward_meter_mean": 0.6031774878501892, "reward_meter_std": 0.3062264323234558, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9900212287902832, "reward_repeat_soft_std": 0.011030333116650581, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.2499571591615677, "reward_total_composite_mean": 0.6786819696426392, "reward_total_composite_std": 0.13810473680496216} {"timestamp_utc": "2026-04-13T02:17:14Z", "mode": "train", "global_step": 1588, "epoch": 0.15951783023606228, "loss": 0.0439, "grad_norm": 21.06489372253418, "learning_rate": 5.190909090909091e-06, "num_tokens": 2859888.0, "completions/mean_length": 28.625, "completions/min_length": 25.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.625, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.5416260957717896, "rewards/meter/std": 0.4434306025505066, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.655731737613678, "rewards/total_composite/std": 0.21951690316200256, "reward": 0.655731737613678, "reward_std": 0.21951690316200256, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17053762078285217, "sampling/sampling_logp_difference/max": 1.6838436126708984, "sampling/importance_sampling_ratio/min": 0.1856590062379837, "sampling/importance_sampling_ratio/mean": 0.9908615350723267, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9443041011691093, "clip_ratio/low_mean": 0.05189860425889492, "clip_ratio/low_min": 0.05189860425889492, "clip_ratio/high_mean": 0.13220734242349863, "clip_ratio/high_max": 0.13220734242349863, "clip_ratio/region_mean": 0.18410594668239355, "reward_total_mean": 0.655731737613678, "reward_meter_mean": 0.5416260957717896, "reward_meter_std": 0.4434306025505066, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.655731737613678, "reward_total_composite_std": 0.21951690316200256} {"timestamp_utc": "2026-04-13T02:17:21Z", "mode": "train", "global_step": 1589, "epoch": 0.15961828227021596, "loss": 0.0332, "grad_norm": 16.670177459716797, "learning_rate": 5.187878787878788e-06, "num_tokens": 2861500.0, "completions/mean_length": 54.5, "completions/min_length": 46.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.5, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.6491508483886719, "rewards/meter/std": 0.373673677444458, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9832876920700073, "rewards/repeat_soft/std": 0.016959769651293755, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.13265828788280487, "rewards/total_composite/mean": 0.6923216581344604, "rewards/total_composite/std": 0.14860621094703674, "reward": 0.6923216581344604, "reward_std": 0.14860619604587555, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13605745136737823, "sampling/sampling_logp_difference/max": 2.3364038467407227, "sampling/importance_sampling_ratio/min": 0.09667467325925827, "sampling/importance_sampling_ratio/mean": 1.005554437637329, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8266329765319824, "clip_ratio/low_mean": 0.07274613995105028, "clip_ratio/low_min": 0.07274613995105028, "clip_ratio/high_mean": 0.05844948347657919, "clip_ratio/high_max": 0.05844948347657919, "clip_ratio/region_mean": 0.13119562342762947, "reward_total_mean": 0.6923216581344604, "reward_meter_mean": 0.6491508483886719, "reward_meter_std": 0.373673677444458, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9832876920700073, "reward_repeat_soft_std": 0.016959769651293755, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.13265828788280487, "reward_total_composite_mean": 0.6923216581344604, "reward_total_composite_std": 0.14860621094703674} {"timestamp_utc": "2026-04-13T02:17:28Z", "mode": "train", "global_step": 1590, "epoch": 0.15971873430436967, "loss": 0.0174, "grad_norm": 8.030685424804688, "learning_rate": 5.184848484848485e-06, "num_tokens": 2863545.0, "completions/mean_length": 96.625, "completions/min_length": 82.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.625, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.9286532998085022, "rewards/meter/std": 0.09843289852142334, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9790492653846741, "rewards/repeat_soft/std": 0.020258894190192223, "rewards/judge_quality/mean": 0.5324999690055847, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.8255489468574524, "rewards/total_composite/std": 0.07479030638933182, "reward": 0.8255489468574524, "reward_std": 0.07479028403759003, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11467957496643066, "sampling/sampling_logp_difference/max": 1.7478208541870117, "sampling/importance_sampling_ratio/min": 0.17415302991867065, "sampling/importance_sampling_ratio/mean": 0.9970564246177673, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6829348132014275, "clip_ratio/low_mean": 0.05464051617309451, "clip_ratio/low_min": 0.05464051617309451, "clip_ratio/high_mean": 0.04207002650946379, "clip_ratio/high_max": 0.04207002650946379, "clip_ratio/region_mean": 0.0967105426825583, "reward_total_mean": 0.8255489468574524, "reward_meter_mean": 0.9286532998085022, "reward_meter_std": 0.09843289852142334, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9790492653846741, "reward_repeat_soft_std": 0.020258894190192223, "reward_judge_quality_mean": 0.5324999690055847, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.8255489468574524, "reward_total_composite_std": 0.07479030638933182} {"timestamp_utc": "2026-04-13T02:17:35Z", "mode": "train", "global_step": 1591, "epoch": 0.15981918633852335, "loss": 0.0289, "grad_norm": 22.17462730407715, "learning_rate": 5.181818181818182e-06, "num_tokens": 2864988.0, "completions/mean_length": 29.375, "completions/min_length": 25.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.375, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.8429169654846191, "rewards/meter/std": 0.27223560214042664, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.32499998807907104, "rewards/judge_quality/std": 0.1035098284482956, "rewards/total_composite/mean": 0.7230626344680786, "rewards/total_composite/std": 0.1387884020805359, "reward": 0.7230626344680786, "reward_std": 0.13878841698169708, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1174558699131012, "sampling/sampling_logp_difference/max": 1.5198485851287842, "sampling/importance_sampling_ratio/min": 0.21874502301216125, "sampling/importance_sampling_ratio/mean": 1.0174630880355835, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7656838074326515, "clip_ratio/low_mean": 0.02356902416795492, "clip_ratio/low_min": 0.02356902416795492, "clip_ratio/high_mean": 0.07917462522163987, "clip_ratio/high_max": 0.07917462522163987, "clip_ratio/region_mean": 0.1027436493895948, "reward_total_mean": 0.7230626344680786, "reward_meter_mean": 0.8429169654846191, "reward_meter_std": 0.27223560214042664, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.32499998807907104, "reward_judge_quality_std": 0.1035098284482956, "reward_total_composite_mean": 0.7230626344680786, "reward_total_composite_std": 0.1387884020805359} {"timestamp_utc": "2026-04-13T02:17:41Z", "mode": "train", "global_step": 1592, "epoch": 0.15991963837267706, "loss": 0.0804, "grad_norm": 14.293763160705566, "learning_rate": 5.1787878787878784e-06, "num_tokens": 2866726.0, "completions/mean_length": 52.25, "completions/min_length": 40.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.25, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.7821581959724426, "rewards/meter/std": 0.28935909271240234, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9222293496131897, "rewards/repeat_soft/std": 0.046193014830350876, "rewards/judge_quality/mean": 0.3374999761581421, "rewards/judge_quality/std": 0.12464234232902527, "rewards/total_composite/mean": 0.6954441070556641, "rewards/total_composite/std": 0.13994547724723816, "reward": 0.6954441070556641, "reward_std": 0.13994544744491577, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14297328889369965, "sampling/sampling_logp_difference/max": 1.8713569641113281, "sampling/importance_sampling_ratio/min": 0.153914675116539, "sampling/importance_sampling_ratio/mean": 1.010843276977539, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8762386068701744, "clip_ratio/low_mean": 0.047355770133435726, "clip_ratio/low_min": 0.047355770133435726, "clip_ratio/high_mean": 0.1011978080496192, "clip_ratio/high_max": 0.1011978080496192, "clip_ratio/region_mean": 0.14855357818305492, "reward_total_mean": 0.6954441070556641, "reward_meter_mean": 0.7821581959724426, "reward_meter_std": 0.28935909271240234, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9222293496131897, "reward_repeat_soft_std": 0.046193014830350876, "reward_judge_quality_mean": 0.3374999761581421, "reward_judge_quality_std": 0.12464234232902527, "reward_total_composite_mean": 0.6954441070556641, "reward_total_composite_std": 0.13994547724723816} {"timestamp_utc": "2026-04-13T02:17:48Z", "mode": "train", "global_step": 1593, "epoch": 0.16002009040683074, "loss": 0.0745, "grad_norm": 13.93443489074707, "learning_rate": 5.1757575757575765e-06, "num_tokens": 2868450.0, "completions/mean_length": 55.5, "completions/min_length": 46.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.5, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.6624391674995422, "rewards/meter/std": 0.3811221718788147, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9934958219528198, "rewards/repeat_soft/std": 0.005858998280018568, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.10260014235973358, "rewards/total_composite/mean": 0.6880722045898438, "rewards/total_composite/std": 0.18167757987976074, "reward": 0.6880722045898438, "reward_std": 0.18167759478092194, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15146559476852417, "sampling/sampling_logp_difference/max": 1.8455657958984375, "sampling/importance_sampling_ratio/min": 0.15793593227863312, "sampling/importance_sampling_ratio/mean": 1.031455636024475, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1415689513087273, "clip_ratio/low_mean": 0.04967669118195772, "clip_ratio/low_min": 0.04967669118195772, "clip_ratio/high_mean": 0.0864011300727725, "clip_ratio/high_max": 0.0864011300727725, "clip_ratio/region_mean": 0.13607782125473022, "reward_total_mean": 0.6880722045898438, "reward_meter_mean": 0.6624391674995422, "reward_meter_std": 0.3811221718788147, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9934958219528198, "reward_repeat_soft_std": 0.005858998280018568, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.10260014235973358, "reward_total_composite_mean": 0.6880722045898438, "reward_total_composite_std": 0.18167757987976074} {"timestamp_utc": "2026-04-13T02:17:56Z", "mode": "train", "global_step": 1594, "epoch": 0.16012054244098442, "loss": 0.0034, "grad_norm": 7.079912185668945, "learning_rate": 5.172727272727273e-06, "num_tokens": 2871236.0, "completions/mean_length": 136.25, "completions/min_length": 111.0, "completions/max_length": 157.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.25, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.530052900314331, "rewards/meter/std": 0.3439391255378723, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9414882659912109, "rewards/repeat_soft/std": 0.031143339350819588, "rewards/judge_quality/mean": 0.3062500059604645, "rewards/judge_quality/std": 0.1293872892856598, "rewards/total_composite/mean": 0.5745476484298706, "rewards/total_composite/std": 0.17273679375648499, "reward": 0.5745476484298706, "reward_std": 0.1727367639541626, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12907789647579193, "sampling/sampling_logp_difference/max": 1.8719110488891602, "sampling/importance_sampling_ratio/min": 0.1538294106721878, "sampling/importance_sampling_ratio/mean": 1.0222257375717163, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9681661427021027, "clip_ratio/low_mean": 0.06396698160097003, "clip_ratio/low_min": 0.06396698160097003, "clip_ratio/high_mean": 0.04335795808583498, "clip_ratio/high_max": 0.04335795808583498, "clip_ratio/region_mean": 0.10732493968680501, "reward_total_mean": 0.5745476484298706, "reward_meter_mean": 0.530052900314331, "reward_meter_std": 0.3439391255378723, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9414882659912109, "reward_repeat_soft_std": 0.031143339350819588, "reward_judge_quality_mean": 0.3062500059604645, "reward_judge_quality_std": 0.1293872892856598, "reward_total_composite_mean": 0.5745476484298706, "reward_total_composite_std": 0.17273679375648499} {"timestamp_utc": "2026-04-13T02:18:03Z", "mode": "train", "global_step": 1595, "epoch": 0.16022099447513813, "loss": -0.0122, "grad_norm": 18.019779205322266, "learning_rate": 5.16969696969697e-06, "num_tokens": 2872898.0, "completions/mean_length": 38.75, "completions/min_length": 35.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.75, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.7600524425506592, "rewards/meter/std": 0.3190794587135315, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9972118735313416, "rewards/repeat_soft/std": 0.004235220141708851, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.7398698329925537, "rewards/total_composite/std": 0.14834223687648773, "reward": 0.7398698329925537, "reward_std": 0.14834223687648773, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11918831616640091, "sampling/sampling_logp_difference/max": 1.4503841400146484, "sampling/importance_sampling_ratio/min": 0.23448020219802856, "sampling/importance_sampling_ratio/mean": 0.9964276552200317, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6210056059062481, "clip_ratio/low_mean": 0.03344298247247934, "clip_ratio/low_min": 0.03344298247247934, "clip_ratio/high_mean": 0.08518184581771493, "clip_ratio/high_max": 0.08518184581771493, "clip_ratio/region_mean": 0.11862482829019427, "reward_total_mean": 0.7398698329925537, "reward_meter_mean": 0.7600524425506592, "reward_meter_std": 0.3190794587135315, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9972118735313416, "reward_repeat_soft_std": 0.004235220141708851, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.7398698329925537, "reward_total_composite_std": 0.14834223687648773} {"timestamp_utc": "2026-04-13T02:18:11Z", "mode": "train", "global_step": 1596, "epoch": 0.1603214465092918, "loss": 0.0134, "grad_norm": 6.842519283294678, "learning_rate": 5.1666666666666675e-06, "num_tokens": 2875804.0, "completions/mean_length": 158.25, "completions/min_length": 145.0, "completions/max_length": 176.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 158.25, "completions/min_terminated_length": 145.0, "completions/max_terminated_length": 176.0, "rewards/meter/mean": 0.910365641117096, "rewards/meter/std": 0.1903901845216751, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.843002438545227, "rewards/repeat_soft/std": 0.08399877697229385, "rewards/judge_quality/mean": 0.26749998331069946, "rewards/judge_quality/std": 0.10375107079744339, "rewards/total_composite/mean": 0.7242147922515869, "rewards/total_composite/std": 0.06809654086828232, "reward": 0.7242147922515869, "reward_std": 0.06809654831886292, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10769176483154297, "sampling/sampling_logp_difference/max": 2.1974925994873047, "sampling/importance_sampling_ratio/min": 0.11108133941888809, "sampling/importance_sampling_ratio/mean": 1.0104340314865112, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6299064606428146, "clip_ratio/low_mean": 0.019343991298228502, "clip_ratio/low_min": 0.019343991298228502, "clip_ratio/high_mean": 0.07677728682756424, "clip_ratio/high_max": 0.07677728682756424, "clip_ratio/region_mean": 0.09612127812579274, "reward_total_mean": 0.7242147922515869, "reward_meter_mean": 0.910365641117096, "reward_meter_std": 0.1903901845216751, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.843002438545227, "reward_repeat_soft_std": 0.08399877697229385, "reward_judge_quality_mean": 0.26749998331069946, "reward_judge_quality_std": 0.10375107079744339, "reward_total_composite_mean": 0.7242147922515869, "reward_total_composite_std": 0.06809654086828232} {"timestamp_utc": "2026-04-13T02:18:18Z", "mode": "train", "global_step": 1597, "epoch": 0.16042189854344552, "loss": 0.0122, "grad_norm": 6.189230918884277, "learning_rate": 5.163636363636364e-06, "num_tokens": 2878264.0, "completions/mean_length": 150.5, "completions/min_length": 125.0, "completions/max_length": 172.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 150.5, "completions/min_terminated_length": 125.0, "completions/max_terminated_length": 172.0, "rewards/meter/mean": 0.9040932059288025, "rewards/meter/std": 0.12258608639240265, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9198053479194641, "rewards/repeat_soft/std": 0.05094794183969498, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.7823224067687988, "rewards/total_composite/std": 0.07159575819969177, "reward": 0.7823224067687988, "reward_std": 0.07159575074911118, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12292990833520889, "sampling/sampling_logp_difference/max": 3.0509233474731445, "sampling/importance_sampling_ratio/min": 0.047315217554569244, "sampling/importance_sampling_ratio/mean": 1.014138102531433, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7889858484268188, "clip_ratio/low_mean": 0.040767903439700603, "clip_ratio/low_min": 0.040767903439700603, "clip_ratio/high_mean": 0.07380883023142815, "clip_ratio/high_max": 0.07380883023142815, "clip_ratio/region_mean": 0.11457673367112875, "reward_total_mean": 0.7823224067687988, "reward_meter_mean": 0.9040932059288025, "reward_meter_std": 0.12258608639240265, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9198053479194641, "reward_repeat_soft_std": 0.05094794183969498, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.7823224067687988, "reward_total_composite_std": 0.07159575819969177} {"timestamp_utc": "2026-04-13T02:18:25Z", "mode": "train", "global_step": 1598, "epoch": 0.1605223505775992, "loss": 0.0633, "grad_norm": 13.148232460021973, "learning_rate": 5.160606060606061e-06, "num_tokens": 2879773.0, "completions/mean_length": 47.625, "completions/min_length": 41.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.625, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.8658472299575806, "rewards/meter/std": 0.3306692838668823, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9685888290405273, "rewards/repeat_soft/std": 0.03501502051949501, "rewards/judge_quality/mean": 0.543749988079071, "rewards/judge_quality/std": 0.14647649228572845, "rewards/total_composite/mean": 0.7996151447296143, "rewards/total_composite/std": 0.16771911084651947, "reward": 0.7996151447296143, "reward_std": 0.16771911084651947, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13697229325771332, "sampling/sampling_logp_difference/max": 2.26168155670166, "sampling/importance_sampling_ratio/min": 0.1041751578450203, "sampling/importance_sampling_ratio/mean": 1.0162564516067505, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6973331198096275, "clip_ratio/low_mean": 0.02358490601181984, "clip_ratio/low_min": 0.02358490601181984, "clip_ratio/high_mean": 0.12269435077905655, "clip_ratio/high_max": 0.12269435077905655, "clip_ratio/region_mean": 0.1462792567908764, "reward_total_mean": 0.7996151447296143, "reward_meter_mean": 0.8658472299575806, "reward_meter_std": 0.3306692838668823, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9685888290405273, "reward_repeat_soft_std": 0.03501502051949501, "reward_judge_quality_mean": 0.543749988079071, "reward_judge_quality_std": 0.14647649228572845, "reward_total_composite_mean": 0.7996151447296143, "reward_total_composite_std": 0.16771911084651947} {"timestamp_utc": "2026-04-13T02:18:32Z", "mode": "train", "global_step": 1599, "epoch": 0.16062280261175288, "loss": -0.0173, "grad_norm": 8.519590377807617, "learning_rate": 5.1575757575757575e-06, "num_tokens": 2881681.0, "completions/mean_length": 85.5, "completions/min_length": 75.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.5, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.7790887355804443, "rewards/meter/std": 0.280805766582489, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9895375967025757, "rewards/repeat_soft/std": 0.008573921397328377, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.13695022463798523, "rewards/total_composite/mean": 0.7491687536239624, "rewards/total_composite/std": 0.1368962526321411, "reward": 0.7491687536239624, "reward_std": 0.13689623773097992, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1583624631166458, "sampling/sampling_logp_difference/max": 1.6459274291992188, "sampling/importance_sampling_ratio/min": 0.19283363223075867, "sampling/importance_sampling_ratio/mean": 1.0064350366592407, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8646031022071838, "clip_ratio/low_mean": 0.039880952797830105, "clip_ratio/low_min": 0.039880952797830105, "clip_ratio/high_mean": 0.10522342566400766, "clip_ratio/high_max": 0.10522342566400766, "clip_ratio/region_mean": 0.14510437846183777, "reward_total_mean": 0.7491687536239624, "reward_meter_mean": 0.7790887355804443, "reward_meter_std": 0.280805766582489, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9895375967025757, "reward_repeat_soft_std": 0.008573921397328377, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.13695022463798523, "reward_total_composite_mean": 0.7491687536239624, "reward_total_composite_std": 0.1368962526321411} {"timestamp_utc": "2026-04-13T02:18:44Z", "mode": "train", "global_step": 1600, "epoch": 0.1607232546459066, "loss": -0.1959, "grad_norm": 2.219871997833252, "learning_rate": 5.154545454545456e-06, "num_tokens": 2884080.0, "completions/mean_length": 167.875, "completions/min_length": 113.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 118.71429443359375, "completions/min_terminated_length": 113.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.8813477158546448, "rewards/meter/std": 0.2648129463195801, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9861940145492554, "rewards/repeat_soft/std": 0.009925522841513157, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.2121320515871048, "rewards/total_composite/mean": 0.7024046182632446, "rewards/total_composite/std": 0.3012552559375763, "reward": 0.7024046182632446, "reward_std": 0.3012552857398987, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13579237461090088, "sampling/sampling_logp_difference/max": 2.015988349914551, "sampling/importance_sampling_ratio/min": 0.13318870961666107, "sampling/importance_sampling_ratio/mean": 1.007947564125061, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7964302226901054, "clip_ratio/low_mean": 0.014008620753884315, "clip_ratio/low_min": 0.014008620753884315, "clip_ratio/high_mean": 0.12297785468399525, "clip_ratio/high_max": 0.12297785468399525, "clip_ratio/region_mean": 0.13698647543787956, "reward_total_mean": 0.7024046182632446, "reward_meter_mean": 0.8813477158546448, "reward_meter_std": 0.2648129463195801, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9861940145492554, "reward_repeat_soft_std": 0.009925522841513157, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.2121320515871048, "reward_total_composite_mean": 0.7024046182632446, "reward_total_composite_std": 0.3012552559375763} {"timestamp_utc": "2026-04-13T02:19:42Z", "mode": "eval", "global_step": 1600, "epoch": 0.1607232546459066, "eval_loss": NaN, "eval_runtime": 57.0876, "eval_samples_per_second": 1.401, "eval_steps_per_second": 0.175, "eval_num_tokens": 2884080.0, "eval_completions/mean_length": 91.5375, "eval_completions/min_length": 37.5, "eval_completions/max_length": 192.6, "eval_completions/clipped_ratio": 0.0125, "eval_completions/mean_terminated_length": 85.93392868041992, "eval_completions/min_terminated_length": 37.5, "eval_completions/max_terminated_length": 152.5, "eval_rewards/meter/mean": 0.8168634712696076, "eval_rewards/meter/std": 0.2613574206829071, "eval_rewards/count_adherence/mean": 0.993749988079071, "eval_rewards/count_adherence/std": 0.01767767183482647, "eval_rewards/hard_gate/mean": 0.9875, "eval_rewards/hard_gate/std": 0.03535533845424652, "eval_rewards/repeat_soft/mean": 0.9499654948711396, "eval_rewards/repeat_soft/std": 0.05685000531375408, "eval_rewards/judge_quality/mean": 0.43962499499320984, "eval_rewards/judge_quality/std": 0.11284855026751757, "eval_rewards/total_composite/mean": 0.7384807884693145, "eval_rewards/total_composite/std": 0.1309248436242342, "eval_reward": 0.7384807884693145, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.06979612484574318, "eval_sampling/sampling_logp_difference/max": 1.120776391029358, "eval_sampling/importance_sampling_ratio/min": 0.33794219642877577, "eval_sampling/importance_sampling_ratio/mean": 1.0167855381965638, "eval_sampling/importance_sampling_ratio/max": 1.410465669631958, "eval_entropy": 0.8133332967758179, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7384807884693145, "eval_reward_meter_mean": 0.8168634712696076, "eval_reward_meter_std": 0.2613574206829071, "eval_reward_count_adherence_mean": 0.993749988079071, "eval_reward_count_adherence_std": 0.01767767183482647, "eval_reward_hard_gate_mean": 0.9875, "eval_reward_hard_gate_std": 0.03535533845424652, "eval_reward_repeat_soft_mean": 0.9499654948711396, "eval_reward_repeat_soft_std": 0.05685000531375408, "eval_reward_judge_quality_mean": 0.43962499499320984, "eval_reward_judge_quality_std": 0.11284855026751757, "eval_reward_total_composite_mean": 0.7384807884693145, "eval_reward_total_composite_std": 0.1309248436242342} {"timestamp_utc": "2026-04-13T02:19:52Z", "mode": "train", "global_step": 1601, "epoch": 0.16082370668006027, "loss": 0.0278, "grad_norm": 10.531673431396484, "learning_rate": 5.151515151515152e-06, "num_tokens": 2886220.0, "completions/mean_length": 97.5, "completions/min_length": 91.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.5, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.8490273952484131, "rewards/meter/std": 0.23978877067565918, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9776014685630798, "rewards/repeat_soft/std": 0.013071739114820957, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.755822479724884, "rewards/total_composite/std": 0.10731709003448486, "reward": 0.755822479724884, "reward_std": 0.10731709003448486, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13124941289424896, "sampling/sampling_logp_difference/max": 1.3365325927734375, "sampling/importance_sampling_ratio/min": 0.2627551853656769, "sampling/importance_sampling_ratio/mean": 1.0053719282150269, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8098358064889908, "clip_ratio/low_mean": 0.02678571455180645, "clip_ratio/low_min": 0.02678571455180645, "clip_ratio/high_mean": 0.13038233201950788, "clip_ratio/high_max": 0.13038233201950788, "clip_ratio/region_mean": 0.15716804657131433, "reward_total_mean": 0.755822479724884, "reward_meter_mean": 0.8490273952484131, "reward_meter_std": 0.23978877067565918, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9776014685630798, "reward_repeat_soft_std": 0.013071739114820957, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.755822479724884, "reward_total_composite_std": 0.10731709003448486} {"timestamp_utc": "2026-04-13T02:19:58Z", "mode": "train", "global_step": 1602, "epoch": 0.16092415871421395, "loss": 0.0293, "grad_norm": 11.375246047973633, "learning_rate": 5.148484848484849e-06, "num_tokens": 2888042.0, "completions/mean_length": 62.75, "completions/min_length": 57.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.75, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9483268857002258, "rewards/meter/std": 0.10014087706804276, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.996786892414093, "rewards/repeat_soft/std": 0.004183145705610514, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.5985167026519775, "rewards/total_composite/std": 0.3720400035381317, "reward": 0.5985167026519775, "reward_std": 0.3720399737358093, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14833439886569977, "sampling/sampling_logp_difference/max": 1.554276943206787, "sampling/importance_sampling_ratio/min": 0.21134212613105774, "sampling/importance_sampling_ratio/mean": 1.0173770189285278, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1050898060202599, "clip_ratio/low_mean": 0.026814516633749008, "clip_ratio/low_min": 0.026814516633749008, "clip_ratio/high_mean": 0.1043073758482933, "clip_ratio/high_max": 0.1043073758482933, "clip_ratio/region_mean": 0.1311218924820423, "reward_total_mean": 0.5985167026519775, "reward_meter_mean": 0.9483268857002258, "reward_meter_std": 0.10014087706804276, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.996786892414093, "reward_repeat_soft_std": 0.004183145705610514, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.5985167026519775, "reward_total_composite_std": 0.3720400035381317} {"timestamp_utc": "2026-04-13T02:20:06Z", "mode": "train", "global_step": 1603, "epoch": 0.16102461074836766, "loss": 0.0596, "grad_norm": 13.13692855834961, "learning_rate": 5.145454545454546e-06, "num_tokens": 2889925.0, "completions/mean_length": 53.375, "completions/min_length": 50.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.375, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.8126786351203918, "rewards/meter/std": 0.3128056228160858, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9762141108512878, "rewards/repeat_soft/std": 0.018263747915625572, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.7580767869949341, "rewards/total_composite/std": 0.16029895842075348, "reward": 0.7580767869949341, "reward_std": 0.16029894351959229, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1636873185634613, "sampling/sampling_logp_difference/max": 4.540555000305176, "sampling/importance_sampling_ratio/min": 0.010667485184967518, "sampling/importance_sampling_ratio/mean": 1.0043047666549683, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8879844173789024, "clip_ratio/low_mean": 0.03456565085798502, "clip_ratio/low_min": 0.03456565085798502, "clip_ratio/high_mean": 0.11330668907612562, "clip_ratio/high_max": 0.11330668907612562, "clip_ratio/region_mean": 0.14787233993411064, "reward_total_mean": 0.7580767869949341, "reward_meter_mean": 0.8126786351203918, "reward_meter_std": 0.3128056228160858, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9762141108512878, "reward_repeat_soft_std": 0.018263747915625572, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.7580767869949341, "reward_total_composite_std": 0.16029895842075348} {"timestamp_utc": "2026-04-13T02:20:13Z", "mode": "train", "global_step": 1604, "epoch": 0.16112506278252134, "loss": 0.0124, "grad_norm": 16.59724235534668, "learning_rate": 5.142424242424243e-06, "num_tokens": 2891276.0, "completions/mean_length": 23.875, "completions/min_length": 21.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.875, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9819136261940002, "rewards/meter/std": 0.02124580554664135, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.6100000143051147, "rewards/judge_quality/std": 0.2386869192123413, "rewards/total_composite/mean": 0.8711111545562744, "rewards/total_composite/std": 0.06820381432771683, "reward": 0.8711111545562744, "reward_std": 0.06820382177829742, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13213638961315155, "sampling/sampling_logp_difference/max": 1.1549761295318604, "sampling/importance_sampling_ratio/min": 0.3331279456615448, "sampling/importance_sampling_ratio/mean": 1.014060616493225, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7953985258936882, "clip_ratio/low_mean": 0.07472509797662497, "clip_ratio/low_min": 0.07472509797662497, "clip_ratio/high_mean": 0.030214803759008646, "clip_ratio/high_max": 0.030214803759008646, "clip_ratio/region_mean": 0.10493990173563361, "reward_total_mean": 0.8711111545562744, "reward_meter_mean": 0.9819136261940002, "reward_meter_std": 0.02124580554664135, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.6100000143051147, "reward_judge_quality_std": 0.2386869192123413, "reward_total_composite_mean": 0.8711111545562744, "reward_total_composite_std": 0.06820381432771683} {"timestamp_utc": "2026-04-13T02:20:19Z", "mode": "train", "global_step": 1605, "epoch": 0.16122551481667505, "loss": -0.0199, "grad_norm": 12.989702224731445, "learning_rate": 5.139393939393939e-06, "num_tokens": 2892759.0, "completions/mean_length": 31.375, "completions/min_length": 29.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.375, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9869410991668701, "rewards/meter/std": 0.009840622544288635, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9467180371284485, "rewards/repeat_soft/std": 0.04463806748390198, "rewards/judge_quality/mean": 0.42124998569488525, "rewards/judge_quality/std": 0.06998724490404129, "rewards/total_composite/mean": 0.8151702880859375, "rewards/total_composite/std": 0.019379734992980957, "reward": 0.8151702880859375, "reward_std": 0.01937973126769066, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10599944740533829, "sampling/sampling_logp_difference/max": 1.6696162223815918, "sampling/importance_sampling_ratio/min": 0.18831932544708252, "sampling/importance_sampling_ratio/mean": 1.0269742012023926, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7652485445141792, "clip_ratio/low_mean": 0.032392894849181175, "clip_ratio/low_min": 0.032392894849181175, "clip_ratio/high_mean": 0.07959127612411976, "clip_ratio/high_max": 0.07959127612411976, "clip_ratio/region_mean": 0.11198417097330093, "reward_total_mean": 0.8151702880859375, "reward_meter_mean": 0.9869410991668701, "reward_meter_std": 0.009840622544288635, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9467180371284485, "reward_repeat_soft_std": 0.04463806748390198, "reward_judge_quality_mean": 0.42124998569488525, "reward_judge_quality_std": 0.06998724490404129, "reward_total_composite_mean": 0.8151702880859375, "reward_total_composite_std": 0.019379734992980957} {"timestamp_utc": "2026-04-13T02:20:27Z", "mode": "train", "global_step": 1606, "epoch": 0.16132596685082873, "loss": 0.0067, "grad_norm": 11.274660110473633, "learning_rate": 5.1363636363636375e-06, "num_tokens": 2894183.0, "completions/mean_length": 28.0, "completions/min_length": 25.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.0, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.9763247966766357, "rewards/meter/std": 0.014061789028346539, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9495148658752441, "rewards/repeat_soft/std": 0.03557299077510834, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.8039226531982422, "rewards/total_composite/std": 0.016400650143623352, "reward": 0.8039226531982422, "reward_std": 0.016400648280978203, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11240801960229874, "sampling/sampling_logp_difference/max": 1.3140902519226074, "sampling/importance_sampling_ratio/min": 0.2687186598777771, "sampling/importance_sampling_ratio/mean": 1.0093573331832886, "sampling/importance_sampling_ratio/max": 1.6926313638687134, "entropy": 0.8641239553689957, "clip_ratio/low_mean": 0.017241379246115685, "clip_ratio/low_min": 0.017241379246115685, "clip_ratio/high_mean": 0.10954685462638736, "clip_ratio/high_max": 0.10954685462638736, "clip_ratio/region_mean": 0.12678823387250304, "reward_total_mean": 0.8039226531982422, "reward_meter_mean": 0.9763247966766357, "reward_meter_std": 0.014061789028346539, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9495148658752441, "reward_repeat_soft_std": 0.03557299077510834, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.8039226531982422, "reward_total_composite_std": 0.016400650143623352} {"timestamp_utc": "2026-04-13T02:20:33Z", "mode": "train", "global_step": 1607, "epoch": 0.1614264188849824, "loss": 0.0009, "grad_norm": 13.654260635375977, "learning_rate": 5.133333333333334e-06, "num_tokens": 2895719.0, "completions/mean_length": 39.0, "completions/min_length": 36.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.0, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.8332143425941467, "rewards/meter/std": 0.32760855555534363, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9657265543937683, "rewards/repeat_soft/std": 0.032232534140348434, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.7696441411972046, "rewards/total_composite/std": 0.16079366207122803, "reward": 0.7696441411972046, "reward_std": 0.16079366207122803, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09604750573635101, "sampling/sampling_logp_difference/max": 1.7514796257019043, "sampling/importance_sampling_ratio/min": 0.17351700365543365, "sampling/importance_sampling_ratio/mean": 0.9911039471626282, "sampling/importance_sampling_ratio/max": 1.7850472927093506, "entropy": 0.5360786281526089, "clip_ratio/low_mean": 0.023204125929623842, "clip_ratio/low_min": 0.023204125929623842, "clip_ratio/high_mean": 0.09199361503124237, "clip_ratio/high_max": 0.09199361503124237, "clip_ratio/region_mean": 0.11519774096086621, "reward_total_mean": 0.7696441411972046, "reward_meter_mean": 0.8332143425941467, "reward_meter_std": 0.32760855555534363, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9657265543937683, "reward_repeat_soft_std": 0.032232534140348434, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.7696441411972046, "reward_total_composite_std": 0.16079366207122803} {"timestamp_utc": "2026-04-13T02:20:44Z", "mode": "train", "global_step": 1608, "epoch": 0.16152687091913612, "loss": 0.0228, "grad_norm": 6.95980978012085, "learning_rate": 5.130303030303031e-06, "num_tokens": 2898555.0, "completions/mean_length": 157.5, "completions/min_length": 144.0, "completions/max_length": 184.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 157.5, "completions/min_terminated_length": 144.0, "completions/max_terminated_length": 184.0, "rewards/meter/mean": 0.9934732913970947, "rewards/meter/std": 0.005287264008074999, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8848471641540527, "rewards/repeat_soft/std": 0.07640983164310455, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.788672685623169, "rewards/total_composite/std": 0.028153855353593826, "reward": 0.788672685623169, "reward_std": 0.028153857216238976, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13429999351501465, "sampling/sampling_logp_difference/max": 2.1733827590942383, "sampling/importance_sampling_ratio/min": 0.11379203200340271, "sampling/importance_sampling_ratio/mean": 1.0074173212051392, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8698181137442589, "clip_ratio/low_mean": 0.025695225223898888, "clip_ratio/low_min": 0.025695225223898888, "clip_ratio/high_mean": 0.0916399285197258, "clip_ratio/high_max": 0.0916399285197258, "clip_ratio/region_mean": 0.11733515374362469, "reward_total_mean": 0.788672685623169, "reward_meter_mean": 0.9934732913970947, "reward_meter_std": 0.005287264008074999, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8848471641540527, "reward_repeat_soft_std": 0.07640983164310455, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.788672685623169, "reward_total_composite_std": 0.028153855353593826} {"timestamp_utc": "2026-04-13T02:20:51Z", "mode": "train", "global_step": 1609, "epoch": 0.1616273229532898, "loss": -0.0147, "grad_norm": 13.060791015625, "learning_rate": 5.1272727272727275e-06, "num_tokens": 2900284.0, "completions/mean_length": 50.125, "completions/min_length": 44.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.125, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9576990604400635, "rewards/meter/std": 0.0833798423409462, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.964083194732666, "rewards/repeat_soft/std": 0.04344302415847778, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.8078728914260864, "rewards/total_composite/std": 0.03889333829283714, "reward": 0.8078728914260864, "reward_std": 0.03889334201812744, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14040519297122955, "sampling/sampling_logp_difference/max": 1.278085708618164, "sampling/importance_sampling_ratio/min": 0.2785700559616089, "sampling/importance_sampling_ratio/mean": 1.0108274221420288, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9978232160210609, "clip_ratio/low_mean": 0.010638297535479069, "clip_ratio/low_min": 0.010638297535479069, "clip_ratio/high_mean": 0.1262515988200903, "clip_ratio/high_max": 0.1262515988200903, "clip_ratio/region_mean": 0.13688989635556936, "reward_total_mean": 0.8078728914260864, "reward_meter_mean": 0.9576990604400635, "reward_meter_std": 0.0833798423409462, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.964083194732666, "reward_repeat_soft_std": 0.04344302415847778, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.8078728914260864, "reward_total_composite_std": 0.03889333829283714} {"timestamp_utc": "2026-04-13T02:21:00Z", "mode": "train", "global_step": 1610, "epoch": 0.1617277749874435, "loss": -0.0341, "grad_norm": 5.727341651916504, "learning_rate": 5.124242424242425e-06, "num_tokens": 2902998.0, "completions/mean_length": 169.25, "completions/min_length": 147.0, "completions/max_length": 219.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 169.25, "completions/min_terminated_length": 147.0, "completions/max_terminated_length": 219.0, "rewards/meter/mean": 0.9193468689918518, "rewards/meter/std": 0.19880734384059906, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8289355039596558, "rewards/repeat_soft/std": 0.11574573069810867, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7694746255874634, "rewards/total_composite/std": 0.08847551792860031, "reward": 0.7694746255874634, "reward_std": 0.08847550302743912, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11999805271625519, "sampling/sampling_logp_difference/max": 1.8380634784698486, "sampling/importance_sampling_ratio/min": 0.15912528336048126, "sampling/importance_sampling_ratio/mean": 1.0119247436523438, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.798811785876751, "clip_ratio/low_mean": 0.021683674305677414, "clip_ratio/low_min": 0.021683674305677414, "clip_ratio/high_mean": 0.08819961361587048, "clip_ratio/high_max": 0.08819961361587048, "clip_ratio/region_mean": 0.10988328792154789, "reward_total_mean": 0.7694746255874634, "reward_meter_mean": 0.9193468689918518, "reward_meter_std": 0.19880734384059906, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8289355039596558, "reward_repeat_soft_std": 0.11574573069810867, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7694746255874634, "reward_total_composite_std": 0.08847551792860031} {"timestamp_utc": "2026-04-13T02:21:07Z", "mode": "train", "global_step": 1611, "epoch": 0.1618282270215972, "loss": 0.0369, "grad_norm": 14.308511734008789, "learning_rate": 5.121212121212121e-06, "num_tokens": 2904507.0, "completions/mean_length": 37.625, "completions/min_length": 34.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.625, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.7359254360198975, "rewards/meter/std": 0.3800547420978546, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.999825656414032, "rewards/repeat_soft/std": 0.00042824377305805683, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.8008989691734314, "rewards/total_composite/std": 0.14710436761379242, "reward": 0.8008989691734314, "reward_std": 0.1471043825149536, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13087868690490723, "sampling/sampling_logp_difference/max": 1.537519931793213, "sampling/importance_sampling_ratio/min": 0.21491344273090363, "sampling/importance_sampling_ratio/mean": 1.0263022184371948, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.624945517629385, "clip_ratio/low_mean": 0.037162160500884056, "clip_ratio/low_min": 0.037162160500884056, "clip_ratio/high_mean": 0.08499408257193863, "clip_ratio/high_max": 0.08499408257193863, "clip_ratio/region_mean": 0.12215624307282269, "reward_total_mean": 0.8008989691734314, "reward_meter_mean": 0.7359254360198975, "reward_meter_std": 0.3800547420978546, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.999825656414032, "reward_repeat_soft_std": 0.00042824377305805683, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.8008989691734314, "reward_total_composite_std": 0.14710436761379242} {"timestamp_utc": "2026-04-13T02:21:15Z", "mode": "train", "global_step": 1612, "epoch": 0.16192867905575087, "loss": 0.0214, "grad_norm": 7.82056999206543, "learning_rate": 5.1181818181818185e-06, "num_tokens": 2907061.0, "completions/mean_length": 114.25, "completions/min_length": 103.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.25, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.8963587284088135, "rewards/meter/std": 0.25292766094207764, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9750574827194214, "rewards/repeat_soft/std": 0.01767103187739849, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7704921960830688, "rewards/total_composite/std": 0.1123206615447998, "reward": 0.7704921960830688, "reward_std": 0.1123206689953804, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14542309939861298, "sampling/sampling_logp_difference/max": 1.857943058013916, "sampling/importance_sampling_ratio/min": 0.15599316358566284, "sampling/importance_sampling_ratio/mean": 1.0123242139816284, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1150662377476692, "clip_ratio/low_mean": 0.020285087637603283, "clip_ratio/low_min": 0.020285087637603283, "clip_ratio/high_mean": 0.10844546090811491, "clip_ratio/high_max": 0.10844546090811491, "clip_ratio/region_mean": 0.1287305485457182, "reward_total_mean": 0.7704921960830688, "reward_meter_mean": 0.8963587284088135, "reward_meter_std": 0.25292766094207764, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9750574827194214, "reward_repeat_soft_std": 0.01767103187739849, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7704921960830688, "reward_total_composite_std": 0.1123206615447998} {"timestamp_utc": "2026-04-13T02:21:24Z", "mode": "train", "global_step": 1613, "epoch": 0.16202913108990458, "loss": 0.0574, "grad_norm": 6.757297515869141, "learning_rate": 5.115151515151515e-06, "num_tokens": 2910045.0, "completions/mean_length": 169.0, "completions/min_length": 137.0, "completions/max_length": 203.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 169.0, "completions/min_terminated_length": 137.0, "completions/max_terminated_length": 203.0, "rewards/meter/mean": 0.8573089838027954, "rewards/meter/std": 0.2656608521938324, "rewards/count_adherence/mean": 0.925000011920929, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9663785696029663, "rewards/repeat_soft/std": 0.015304680913686752, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.14520922303199768, "rewards/total_composite/mean": 0.7456768751144409, "rewards/total_composite/std": 0.0915573388338089, "reward": 0.7456768751144409, "reward_std": 0.0915573462843895, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1435859501361847, "sampling/sampling_logp_difference/max": 2.565211772918701, "sampling/importance_sampling_ratio/min": 0.07690289616584778, "sampling/importance_sampling_ratio/mean": 1.0301074981689453, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2162883132696152, "clip_ratio/low_mean": 0.049904169514775276, "clip_ratio/low_min": 0.049904169514775276, "clip_ratio/high_mean": 0.08041460812091827, "clip_ratio/high_max": 0.08041460812091827, "clip_ratio/region_mean": 0.13031877763569355, "reward_total_mean": 0.7456768751144409, "reward_meter_mean": 0.8573089838027954, "reward_meter_std": 0.2656608521938324, "reward_count_adherence_mean": 0.925000011920929, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9663785696029663, "reward_repeat_soft_std": 0.015304680913686752, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.14520922303199768, "reward_total_composite_mean": 0.7456768751144409, "reward_total_composite_std": 0.0915573388338089} {"timestamp_utc": "2026-04-13T02:21:31Z", "mode": "train", "global_step": 1614, "epoch": 0.16212958312405826, "loss": 0.0209, "grad_norm": 8.574506759643555, "learning_rate": 5.112121212121213e-06, "num_tokens": 2912179.0, "completions/mean_length": 87.75, "completions/min_length": 77.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.75, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.905223548412323, "rewards/meter/std": 0.1271200180053711, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9172911643981934, "rewards/repeat_soft/std": 0.04112652316689491, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.8050796985626221, "rewards/total_composite/std": 0.09149721264839172, "reward": 0.8050796985626221, "reward_std": 0.09149722009897232, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09046916663646698, "sampling/sampling_logp_difference/max": 1.609813928604126, "sampling/importance_sampling_ratio/min": 0.19992481172084808, "sampling/importance_sampling_ratio/mean": 1.0039575099945068, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45774611085653305, "clip_ratio/low_mean": 0.03327103145420551, "clip_ratio/low_min": 0.03327103145420551, "clip_ratio/high_mean": 0.048207002226263285, "clip_ratio/high_max": 0.048207002226263285, "clip_ratio/region_mean": 0.0814780336804688, "reward_total_mean": 0.8050796985626221, "reward_meter_mean": 0.905223548412323, "reward_meter_std": 0.1271200180053711, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9172911643981934, "reward_repeat_soft_std": 0.04112652316689491, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.8050796985626221, "reward_total_composite_std": 0.09149721264839172} {"timestamp_utc": "2026-04-13T02:21:38Z", "mode": "train", "global_step": 1615, "epoch": 0.16223003515821197, "loss": -0.0282, "grad_norm": 9.967058181762695, "learning_rate": 5.109090909090909e-06, "num_tokens": 2913856.0, "completions/mean_length": 56.625, "completions/min_length": 53.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.625, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.8223339319229126, "rewards/meter/std": 0.3258882164955139, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9695556163787842, "rewards/repeat_soft/std": 0.014594295993447304, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.8180058002471924, "rewards/total_composite/std": 0.15386317670345306, "reward": 0.8180058002471924, "reward_std": 0.15386316180229187, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11338113993406296, "sampling/sampling_logp_difference/max": 1.8059029579162598, "sampling/importance_sampling_ratio/min": 0.164326012134552, "sampling/importance_sampling_ratio/mean": 1.023408055305481, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7534859552979469, "clip_ratio/low_mean": 0.04564264044165611, "clip_ratio/low_min": 0.04564264044165611, "clip_ratio/high_mean": 0.04821833083406091, "clip_ratio/high_max": 0.04821833083406091, "clip_ratio/region_mean": 0.09386097127571702, "reward_total_mean": 0.8180058002471924, "reward_meter_mean": 0.8223339319229126, "reward_meter_std": 0.3258882164955139, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9695556163787842, "reward_repeat_soft_std": 0.014594295993447304, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.8180058002471924, "reward_total_composite_std": 0.15386317670345306} {"timestamp_utc": "2026-04-13T02:21:49Z", "mode": "train", "global_step": 1616, "epoch": 0.16233048719236565, "loss": -0.0943, "grad_norm": 1.7823607921600342, "learning_rate": 5.106060606060607e-06, "num_tokens": 2915312.0, "completions/mean_length": 89.0, "completions/min_length": 23.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 28.571430206298828, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.865065336227417, "rewards/meter/std": 0.3490946888923645, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9569318294525146, "rewards/repeat_soft/std": 0.015749182552099228, "rewards/judge_quality/mean": 0.3687500059604645, "rewards/judge_quality/std": 0.2703668177127838, "rewards/total_composite/mean": 0.7128673791885376, "rewards/total_composite/std": 0.296734482049942, "reward": 0.7128673791885376, "reward_std": 0.296734482049942, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16694898903369904, "sampling/sampling_logp_difference/max": 1.21602201461792, "sampling/importance_sampling_ratio/min": 0.29640692472457886, "sampling/importance_sampling_ratio/mean": 1.0401705503463745, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.354978285729885, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.15752647165209055, "clip_ratio/high_max": 0.15752647165209055, "clip_ratio/region_mean": 0.15752647165209055, "reward_total_mean": 0.7128673791885376, "reward_meter_mean": 0.865065336227417, "reward_meter_std": 0.3490946888923645, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9569318294525146, "reward_repeat_soft_std": 0.015749182552099228, "reward_judge_quality_mean": 0.3687500059604645, "reward_judge_quality_std": 0.2703668177127838, "reward_total_composite_mean": 0.7128673791885376, "reward_total_composite_std": 0.296734482049942} {"timestamp_utc": "2026-04-13T02:21:56Z", "mode": "train", "global_step": 1617, "epoch": 0.16243093922651933, "loss": 0.1188, "grad_norm": 12.287056922912598, "learning_rate": 5.103030303030303e-06, "num_tokens": 2917345.0, "completions/mean_length": 78.125, "completions/min_length": 67.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.125, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.8819071650505066, "rewards/meter/std": 0.3071115016937256, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9555244445800781, "rewards/repeat_soft/std": 0.023199010640382767, "rewards/judge_quality/mean": 0.5699999928474426, "rewards/judge_quality/std": 0.16035676002502441, "rewards/total_composite/mean": 0.8134106397628784, "rewards/total_composite/std": 0.16137979924678802, "reward": 0.8134106397628784, "reward_std": 0.16137981414794922, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13984303176403046, "sampling/sampling_logp_difference/max": 2.2798166275024414, "sampling/importance_sampling_ratio/min": 0.10230296850204468, "sampling/importance_sampling_ratio/mean": 1.0044713020324707, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8122993931174278, "clip_ratio/low_mean": 0.030432257801294327, "clip_ratio/low_min": 0.030432257801294327, "clip_ratio/high_mean": 0.08272749092429876, "clip_ratio/high_max": 0.08272749092429876, "clip_ratio/region_mean": 0.11315974872559309, "reward_total_mean": 0.8134106397628784, "reward_meter_mean": 0.8819071650505066, "reward_meter_std": 0.3071115016937256, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9555244445800781, "reward_repeat_soft_std": 0.023199010640382767, "reward_judge_quality_mean": 0.5699999928474426, "reward_judge_quality_std": 0.16035676002502441, "reward_total_composite_mean": 0.8134106397628784, "reward_total_composite_std": 0.16137979924678802} {"timestamp_utc": "2026-04-13T02:22:02Z", "mode": "train", "global_step": 1618, "epoch": 0.16253139126067304, "loss": 0.1534, "grad_norm": 19.95191764831543, "learning_rate": 5.1e-06, "num_tokens": 2918888.0, "completions/mean_length": 33.875, "completions/min_length": 27.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.875, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.7377681732177734, "rewards/meter/std": 0.3407692015171051, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9904005527496338, "rewards/repeat_soft/std": 0.009691128507256508, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.700660765171051, "rewards/total_composite/std": 0.17011481523513794, "reward": 0.700660765171051, "reward_std": 0.17011478543281555, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13746516406536102, "sampling/sampling_logp_difference/max": 1.3179349899291992, "sampling/importance_sampling_ratio/min": 0.26768752932548523, "sampling/importance_sampling_ratio/mean": 1.019708275794983, "sampling/importance_sampling_ratio/max": 1.711814284324646, "entropy": 1.2512338608503342, "clip_ratio/low_mean": 0.035054720006883144, "clip_ratio/low_min": 0.035054720006883144, "clip_ratio/high_mean": 0.11149549763649702, "clip_ratio/high_max": 0.11149549763649702, "clip_ratio/region_mean": 0.14655021764338017, "reward_total_mean": 0.700660765171051, "reward_meter_mean": 0.7377681732177734, "reward_meter_std": 0.3407692015171051, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9904005527496338, "reward_repeat_soft_std": 0.009691128507256508, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.700660765171051, "reward_total_composite_std": 0.17011481523513794} {"timestamp_utc": "2026-04-13T02:22:10Z", "mode": "train", "global_step": 1619, "epoch": 0.16263184329482672, "loss": 0.0018, "grad_norm": 8.470718383789062, "learning_rate": 5.096969696969697e-06, "num_tokens": 2921361.0, "completions/mean_length": 131.125, "completions/min_length": 116.0, "completions/max_length": 148.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.125, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 148.0, "rewards/meter/mean": 0.7441799640655518, "rewards/meter/std": 0.11757726967334747, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8984763622283936, "rewards/repeat_soft/std": 0.09564844518899918, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.6048097014427185, "rewards/total_composite/std": 0.25097939372062683, "reward": 0.6048097014427185, "reward_std": 0.25097939372062683, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12092279642820358, "sampling/sampling_logp_difference/max": 3.919762372970581, "sampling/importance_sampling_ratio/min": 0.019845809787511826, "sampling/importance_sampling_ratio/mean": 1.0023690462112427, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6411322765052319, "clip_ratio/low_mean": 0.01587301678955555, "clip_ratio/low_min": 0.01587301678955555, "clip_ratio/high_mean": 0.10371167585253716, "clip_ratio/high_max": 0.10371167585253716, "clip_ratio/region_mean": 0.1195846926420927, "reward_total_mean": 0.6048097014427185, "reward_meter_mean": 0.7441799640655518, "reward_meter_std": 0.11757726967334747, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8984763622283936, "reward_repeat_soft_std": 0.09564844518899918, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.6048097014427185, "reward_total_composite_std": 0.25097939372062683} {"timestamp_utc": "2026-04-13T02:22:17Z", "mode": "train", "global_step": 1620, "epoch": 0.16273229532898043, "loss": 0.0777, "grad_norm": 7.88734245300293, "learning_rate": 5.093939393939395e-06, "num_tokens": 2923779.0, "completions/mean_length": 123.25, "completions/min_length": 99.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 123.25, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.6946475505828857, "rewards/meter/std": 0.26610246300697327, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9392514824867249, "rewards/repeat_soft/std": 0.039315976202487946, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.6613290309906006, "rewards/total_composite/std": 0.12334661185741425, "reward": 0.6613290309906006, "reward_std": 0.12334661185741425, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1509086787700653, "sampling/sampling_logp_difference/max": 1.336362600326538, "sampling/importance_sampling_ratio/min": 0.26279982924461365, "sampling/importance_sampling_ratio/mean": 1.0201829671859741, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3982935100793839, "clip_ratio/low_mean": 0.05036432109773159, "clip_ratio/low_min": 0.05036432109773159, "clip_ratio/high_mean": 0.11393790226429701, "clip_ratio/high_max": 0.11393790226429701, "clip_ratio/region_mean": 0.1643022233620286, "reward_total_mean": 0.6613290309906006, "reward_meter_mean": 0.6946475505828857, "reward_meter_std": 0.26610246300697327, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9392514824867249, "reward_repeat_soft_std": 0.039315976202487946, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.6613290309906006, "reward_total_composite_std": 0.12334661185741425} {"timestamp_utc": "2026-04-13T02:22:25Z", "mode": "train", "global_step": 1621, "epoch": 0.1628327473631341, "loss": 0.0345, "grad_norm": 17.859596252441406, "learning_rate": 5.090909090909091e-06, "num_tokens": 2925386.0, "completions/mean_length": 36.875, "completions/min_length": 32.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.875, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.6147595643997192, "rewards/meter/std": 0.2985612154006958, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9779980182647705, "rewards/repeat_soft/std": 0.024525558575987816, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.6736916303634644, "rewards/total_composite/std": 0.12523820996284485, "reward": 0.6736916303634644, "reward_std": 0.12523820996284485, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15223990380764008, "sampling/sampling_logp_difference/max": 1.7617571353912354, "sampling/importance_sampling_ratio/min": 0.17174282670021057, "sampling/importance_sampling_ratio/mean": 0.9848056435585022, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8220832347869873, "clip_ratio/low_mean": 0.0462359965313226, "clip_ratio/low_min": 0.0462359965313226, "clip_ratio/high_mean": 0.10450130514800549, "clip_ratio/high_max": 0.10450130514800549, "clip_ratio/region_mean": 0.15073730167932808, "reward_total_mean": 0.6736916303634644, "reward_meter_mean": 0.6147595643997192, "reward_meter_std": 0.2985612154006958, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9779980182647705, "reward_repeat_soft_std": 0.024525558575987816, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.6736916303634644, "reward_total_composite_std": 0.12523820996284485} {"timestamp_utc": "2026-04-13T02:22:31Z", "mode": "train", "global_step": 1622, "epoch": 0.1629331993972878, "loss": 0.0223, "grad_norm": 11.33670425415039, "learning_rate": 5.0878787878787885e-06, "num_tokens": 2926979.0, "completions/mean_length": 38.125, "completions/min_length": 36.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.125, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9587262272834778, "rewards/meter/std": 0.025110241025686264, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9696418642997742, "rewards/repeat_soft/std": 0.03177116811275482, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.8265160322189331, "rewards/total_composite/std": 0.04147369787096977, "reward": 0.8265160322189331, "reward_std": 0.041473694145679474, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11610446125268936, "sampling/sampling_logp_difference/max": 1.5859107971191406, "sampling/importance_sampling_ratio/min": 0.20476120710372925, "sampling/importance_sampling_ratio/mean": 1.026472806930542, "sampling/importance_sampling_ratio/max": 1.8235528469085693, "entropy": 0.7521592378616333, "clip_ratio/low_mean": 0.11320327408611774, "clip_ratio/low_min": 0.11320327408611774, "clip_ratio/high_mean": 0.01689189113676548, "clip_ratio/high_max": 0.01689189113676548, "clip_ratio/region_mean": 0.13009516522288322, "reward_total_mean": 0.8265160322189331, "reward_meter_mean": 0.9587262272834778, "reward_meter_std": 0.025110241025686264, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9696418642997742, "reward_repeat_soft_std": 0.03177116811275482, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.8265160322189331, "reward_total_composite_std": 0.04147369787096977} {"timestamp_utc": "2026-04-13T02:22:38Z", "mode": "train", "global_step": 1623, "epoch": 0.1630336514314415, "loss": 0.0146, "grad_norm": 16.151519775390625, "learning_rate": 5.084848484848486e-06, "num_tokens": 2928428.0, "completions/mean_length": 34.125, "completions/min_length": 28.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.125, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.45638465881347656, "rewards/meter/std": 0.2522382140159607, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9953687191009521, "rewards/repeat_soft/std": 0.008268176577985287, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.6371599435806274, "rewards/total_composite/std": 0.1450350433588028, "reward": 0.6371599435806274, "reward_std": 0.1450350284576416, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13147377967834473, "sampling/sampling_logp_difference/max": 2.041128158569336, "sampling/importance_sampling_ratio/min": 0.1298820972442627, "sampling/importance_sampling_ratio/mean": 0.9880146384239197, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5546537786722183, "clip_ratio/low_mean": 0.03900560270994902, "clip_ratio/low_min": 0.03900560270994902, "clip_ratio/high_mean": 0.07150578312575817, "clip_ratio/high_max": 0.07150578312575817, "clip_ratio/region_mean": 0.11051138583570719, "reward_total_mean": 0.6371599435806274, "reward_meter_mean": 0.45638465881347656, "reward_meter_std": 0.2522382140159607, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9953687191009521, "reward_repeat_soft_std": 0.008268176577985287, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.6371599435806274, "reward_total_composite_std": 0.1450350433588028} {"timestamp_utc": "2026-04-13T02:22:45Z", "mode": "train", "global_step": 1624, "epoch": 0.16313410346559518, "loss": 0.0277, "grad_norm": 8.453937530517578, "learning_rate": 5.081818181818182e-06, "num_tokens": 2930718.0, "completions/mean_length": 91.25, "completions/min_length": 81.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.25, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9807966351509094, "rewards/meter/std": 0.021970132365822792, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9476754069328308, "rewards/repeat_soft/std": 0.041135381907224655, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.2499571591615677, "rewards/total_composite/mean": 0.8443759679794312, "rewards/total_composite/std": 0.07728081196546555, "reward": 0.8443759679794312, "reward_std": 0.07728081941604614, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1247146874666214, "sampling/sampling_logp_difference/max": 2.7949600219726562, "sampling/importance_sampling_ratio/min": 0.06111731752753258, "sampling/importance_sampling_ratio/mean": 1.0175502300262451, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7639965489506721, "clip_ratio/low_mean": 0.09936751239001751, "clip_ratio/low_min": 0.09936751239001751, "clip_ratio/high_mean": 0.026643991470336914, "clip_ratio/high_max": 0.026643991470336914, "clip_ratio/region_mean": 0.12601150386035442, "reward_total_mean": 0.8443759679794312, "reward_meter_mean": 0.9807966351509094, "reward_meter_std": 0.021970132365822792, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9476754069328308, "reward_repeat_soft_std": 0.041135381907224655, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.2499571591615677, "reward_total_composite_mean": 0.8443759679794312, "reward_total_composite_std": 0.07728081196546555} {"timestamp_utc": "2026-04-13T02:22:57Z", "mode": "train", "global_step": 1625, "epoch": 0.16323455549974886, "loss": -0.0747, "grad_norm": 2.6699485778808594, "learning_rate": 5.078787878787879e-06, "num_tokens": 2932066.0, "completions/mean_length": 84.5, "completions/min_length": 18.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 23.428571701049805, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.7557849884033203, "rewards/meter/std": 0.4038256108760834, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9671874642372131, "rewards/repeat_soft/std": 0.013258260674774647, "rewards/judge_quality/mean": 0.30000001192092896, "rewards/judge_quality/std": 0.16903084516525269, "rewards/total_composite/mean": 0.6436434388160706, "rewards/total_composite/std": 0.2925571799278259, "reward": 0.6436434388160706, "reward_std": 0.2925571799278259, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1699603945016861, "sampling/sampling_logp_difference/max": 1.9893255233764648, "sampling/importance_sampling_ratio/min": 0.13678765296936035, "sampling/importance_sampling_ratio/mean": 1.001029372215271, "sampling/importance_sampling_ratio/max": 1.7537128925323486, "entropy": 1.1442098692059517, "clip_ratio/low_mean": 0.019999999552965164, "clip_ratio/low_min": 0.019999999552965164, "clip_ratio/high_mean": 0.1716889888048172, "clip_ratio/high_max": 0.1716889888048172, "clip_ratio/region_mean": 0.19168898835778236, "reward_total_mean": 0.6436434388160706, "reward_meter_mean": 0.7557849884033203, "reward_meter_std": 0.4038256108760834, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9671874642372131, "reward_repeat_soft_std": 0.013258260674774647, "reward_judge_quality_mean": 0.30000001192092896, "reward_judge_quality_std": 0.16903084516525269, "reward_total_composite_mean": 0.6436434388160706, "reward_total_composite_std": 0.2925571799278259} {"timestamp_utc": "2026-04-13T02:23:05Z", "mode": "train", "global_step": 1626, "epoch": 0.16333500753390257, "loss": 0.0159, "grad_norm": 9.000627517700195, "learning_rate": 5.075757575757576e-06, "num_tokens": 2934150.0, "completions/mean_length": 86.5, "completions/min_length": 79.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.5, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.9703993797302246, "rewards/meter/std": 0.02066083252429962, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7887475490570068, "rewards/repeat_soft/std": 0.12929682433605194, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7915544509887695, "rewards/total_composite/std": 0.013883921317756176, "reward": 0.7915544509887695, "reward_std": 0.013883916661143303, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09625322371721268, "sampling/sampling_logp_difference/max": 1.323331594467163, "sampling/importance_sampling_ratio/min": 0.27219510078430176, "sampling/importance_sampling_ratio/mean": 1.007745385169983, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5650727041065693, "clip_ratio/low_mean": 0.03261731192469597, "clip_ratio/low_min": 0.03261731192469597, "clip_ratio/high_mean": 0.037360211834311485, "clip_ratio/high_max": 0.037360211834311485, "clip_ratio/region_mean": 0.06997752375900745, "reward_total_mean": 0.7915544509887695, "reward_meter_mean": 0.9703993797302246, "reward_meter_std": 0.02066083252429962, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7887475490570068, "reward_repeat_soft_std": 0.12929682433605194, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7915544509887695, "reward_total_composite_std": 0.013883921317756176} {"timestamp_utc": "2026-04-13T02:23:12Z", "mode": "train", "global_step": 1627, "epoch": 0.16343545956805625, "loss": 0.0309, "grad_norm": 11.827003479003906, "learning_rate": 5.072727272727274e-06, "num_tokens": 2935949.0, "completions/mean_length": 54.875, "completions/min_length": 33.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9232945442199707, "rewards/meter/std": 0.09106792509555817, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9660667777061462, "rewards/repeat_soft/std": 0.029012342914938927, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.811339259147644, "rewards/total_composite/std": 0.06806787848472595, "reward": 0.811339259147644, "reward_std": 0.06806787848472595, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1376182585954666, "sampling/sampling_logp_difference/max": 1.313960075378418, "sampling/importance_sampling_ratio/min": 0.26875367760658264, "sampling/importance_sampling_ratio/mean": 1.0146558284759521, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1275985315442085, "clip_ratio/low_mean": 0.07189476676285267, "clip_ratio/low_min": 0.07189476676285267, "clip_ratio/high_mean": 0.06344618368893862, "clip_ratio/high_max": 0.06344618368893862, "clip_ratio/region_mean": 0.1353409504517913, "reward_total_mean": 0.811339259147644, "reward_meter_mean": 0.9232945442199707, "reward_meter_std": 0.09106792509555817, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9660667777061462, "reward_repeat_soft_std": 0.029012342914938927, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.811339259147644, "reward_total_composite_std": 0.06806787848472595} {"timestamp_utc": "2026-04-13T02:23:18Z", "mode": "train", "global_step": 1628, "epoch": 0.16353591160220995, "loss": 0.0458, "grad_norm": 11.13365364074707, "learning_rate": 5.06969696969697e-06, "num_tokens": 2937585.0, "completions/mean_length": 58.5, "completions/min_length": 55.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.5, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.7724518179893494, "rewards/meter/std": 0.28827980160713196, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.983440637588501, "rewards/repeat_soft/std": 0.01112666167318821, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.7429473996162415, "rewards/total_composite/std": 0.12338326871395111, "reward": 0.7429473996162415, "reward_std": 0.12338326126337051, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15619930624961853, "sampling/sampling_logp_difference/max": 1.7277917861938477, "sampling/importance_sampling_ratio/min": 0.17767632007598877, "sampling/importance_sampling_ratio/mean": 1.0166032314300537, "sampling/importance_sampling_ratio/max": 1.8347948789596558, "entropy": 1.2809825390577316, "clip_ratio/low_mean": 0.03406762331724167, "clip_ratio/low_min": 0.03406762331724167, "clip_ratio/high_mean": 0.12889205012470484, "clip_ratio/high_max": 0.12889205012470484, "clip_ratio/region_mean": 0.1629596734419465, "reward_total_mean": 0.7429473996162415, "reward_meter_mean": 0.7724518179893494, "reward_meter_std": 0.28827980160713196, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.983440637588501, "reward_repeat_soft_std": 0.01112666167318821, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.7429473996162415, "reward_total_composite_std": 0.12338326871395111} {"timestamp_utc": "2026-04-13T02:23:25Z", "mode": "train", "global_step": 1629, "epoch": 0.16363636363636364, "loss": 0.0116, "grad_norm": 9.258278846740723, "learning_rate": 5.0666666666666676e-06, "num_tokens": 2939186.0, "completions/mean_length": 49.125, "completions/min_length": 46.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.125, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.7487030029296875, "rewards/meter/std": 0.4251374900341034, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9496411085128784, "rewards/repeat_soft/std": 0.11640956252813339, "rewards/judge_quality/mean": 0.5362499952316284, "rewards/judge_quality/std": 0.1524970978498459, "rewards/total_composite/mean": 0.7427554130554199, "rewards/total_composite/std": 0.1584431380033493, "reward": 0.7427554130554199, "reward_std": 0.1584431380033493, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1068776473402977, "sampling/sampling_logp_difference/max": 1.32696533203125, "sampling/importance_sampling_ratio/min": 0.2652811110019684, "sampling/importance_sampling_ratio/mean": 1.0204358100891113, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6412968076765537, "clip_ratio/low_mean": 0.025722789578139782, "clip_ratio/low_min": 0.025722789578139782, "clip_ratio/high_mean": 0.09365254081785679, "clip_ratio/high_max": 0.09365254081785679, "clip_ratio/region_mean": 0.11937533039599657, "reward_total_mean": 0.7427554130554199, "reward_meter_mean": 0.7487030029296875, "reward_meter_std": 0.4251374900341034, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9496411085128784, "reward_repeat_soft_std": 0.11640956252813339, "reward_judge_quality_mean": 0.5362499952316284, "reward_judge_quality_std": 0.1524970978498459, "reward_total_composite_mean": 0.7427554130554199, "reward_total_composite_std": 0.1584431380033493} {"timestamp_utc": "2026-04-13T02:23:32Z", "mode": "train", "global_step": 1630, "epoch": 0.16373681567051732, "loss": 0.0876, "grad_norm": 22.766277313232422, "learning_rate": 5.063636363636364e-06, "num_tokens": 2940545.0, "completions/mean_length": 25.875, "completions/min_length": 23.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.875, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.8482903242111206, "rewards/meter/std": 0.33648374676704407, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.7772306203842163, "rewards/total_composite/std": 0.16977441310882568, "reward": 0.7772306203842163, "reward_std": 0.16977442800998688, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14697232842445374, "sampling/sampling_logp_difference/max": 1.3504939079284668, "sampling/importance_sampling_ratio/min": 0.25911223888397217, "sampling/importance_sampling_ratio/mean": 0.9922190308570862, "sampling/importance_sampling_ratio/max": 1.530319333076477, "entropy": 0.8860580772161484, "clip_ratio/low_mean": 0.030092593282461166, "clip_ratio/low_min": 0.030092593282461166, "clip_ratio/high_mean": 0.13399990741163492, "clip_ratio/high_max": 0.13399990741163492, "clip_ratio/region_mean": 0.1640925006940961, "reward_total_mean": 0.7772306203842163, "reward_meter_mean": 0.8482903242111206, "reward_meter_std": 0.33648374676704407, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.7772306203842163, "reward_total_composite_std": 0.16977441310882568} {"timestamp_utc": "2026-04-13T02:23:38Z", "mode": "train", "global_step": 1631, "epoch": 0.16383726770467102, "loss": 0.0493, "grad_norm": 17.864110946655273, "learning_rate": 5.060606060606061e-06, "num_tokens": 2942308.0, "completions/mean_length": 38.375, "completions/min_length": 36.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.375, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.6981111764907837, "rewards/meter/std": 0.3330667018890381, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.973059356212616, "rewards/repeat_soft/std": 0.03595054894685745, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.7062059640884399, "rewards/total_composite/std": 0.11680173128843307, "reward": 0.7062059640884399, "reward_std": 0.11680173128843307, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12074682116508484, "sampling/sampling_logp_difference/max": 1.585233211517334, "sampling/importance_sampling_ratio/min": 0.20489999651908875, "sampling/importance_sampling_ratio/mean": 1.0163826942443848, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.704260803759098, "clip_ratio/low_mean": 0.0674457997083664, "clip_ratio/low_min": 0.0674457997083664, "clip_ratio/high_mean": 0.08843493368476629, "clip_ratio/high_max": 0.08843493368476629, "clip_ratio/region_mean": 0.1558807333931327, "reward_total_mean": 0.7062059640884399, "reward_meter_mean": 0.6981111764907837, "reward_meter_std": 0.3330667018890381, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.973059356212616, "reward_repeat_soft_std": 0.03595054894685745, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.7062059640884399, "reward_total_composite_std": 0.11680173128843307} {"timestamp_utc": "2026-04-13T02:23:47Z", "mode": "train", "global_step": 1632, "epoch": 0.1639377197388247, "loss": -0.0501, "grad_norm": 18.99824333190918, "learning_rate": 5.057575757575758e-06, "num_tokens": 2944048.0, "completions/mean_length": 45.5, "completions/min_length": 36.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.844104528427124, "rewards/meter/std": 0.332962304353714, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9176474809646606, "rewards/repeat_soft/std": 0.055786240845918655, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.7674868106842041, "rewards/total_composite/std": 0.16175395250320435, "reward": 0.7674868106842041, "reward_std": 0.16175396740436554, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15226081013679504, "sampling/sampling_logp_difference/max": 1.7015714645385742, "sampling/importance_sampling_ratio/min": 0.18239666521549225, "sampling/importance_sampling_ratio/mean": 1.0145138502120972, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0189061611890793, "clip_ratio/low_mean": 0.010416666977107525, "clip_ratio/low_min": 0.010416666977107525, "clip_ratio/high_mean": 0.12459492404013872, "clip_ratio/high_max": 0.12459492404013872, "clip_ratio/region_mean": 0.13501159101724625, "reward_total_mean": 0.7674868106842041, "reward_meter_mean": 0.844104528427124, "reward_meter_std": 0.332962304353714, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9176474809646606, "reward_repeat_soft_std": 0.055786240845918655, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.7674868106842041, "reward_total_composite_std": 0.16175395250320435} {"timestamp_utc": "2026-04-13T02:23:54Z", "mode": "train", "global_step": 1633, "epoch": 0.16403817177297841, "loss": 0.01, "grad_norm": 10.67685604095459, "learning_rate": 5.054545454545455e-06, "num_tokens": 2945664.0, "completions/mean_length": 55.0, "completions/min_length": 52.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.0, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.8970602750778198, "rewards/meter/std": 0.10732581466436386, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9947725534439087, "rewards/repeat_soft/std": 0.0070680235512554646, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7791544198989868, "rewards/total_composite/std": 0.04808180406689644, "reward": 0.7791544198989868, "reward_std": 0.04808179661631584, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.134378120303154, "sampling/sampling_logp_difference/max": 1.3192906379699707, "sampling/importance_sampling_ratio/min": 0.3505372405052185, "sampling/importance_sampling_ratio/mean": 1.022526502609253, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1542696580290794, "clip_ratio/low_mean": 0.037478834856301546, "clip_ratio/low_min": 0.037478834856301546, "clip_ratio/high_mean": 0.07805253006517887, "clip_ratio/high_max": 0.07805253006517887, "clip_ratio/region_mean": 0.11553136492148042, "reward_total_mean": 0.7791544198989868, "reward_meter_mean": 0.8970602750778198, "reward_meter_std": 0.10732581466436386, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9947725534439087, "reward_repeat_soft_std": 0.0070680235512554646, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7791544198989868, "reward_total_composite_std": 0.04808180406689644} {"timestamp_utc": "2026-04-13T02:24:06Z", "mode": "train", "global_step": 1634, "epoch": 0.1641386238071321, "loss": -0.1342, "grad_norm": 3.7449142932891846, "learning_rate": 5.051515151515151e-06, "num_tokens": 2947280.0, "completions/mean_length": 120.0, "completions/min_length": 59.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.6855431199073792, "rewards/meter/std": 0.38864585757255554, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9865488409996033, "rewards/repeat_soft/std": 0.01929342746734619, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.6535242795944214, "rewards/total_composite/std": 0.22047758102416992, "reward": 0.6535242795944214, "reward_std": 0.22047758102416992, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15178640186786652, "sampling/sampling_logp_difference/max": 0.9784727096557617, "sampling/importance_sampling_ratio/min": 0.3758847415447235, "sampling/importance_sampling_ratio/mean": 1.0226595401763916, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1445752084255219, "clip_ratio/low_mean": 0.016393441706895828, "clip_ratio/low_min": 0.016393441706895828, "clip_ratio/high_mean": 0.1373176844790578, "clip_ratio/high_max": 0.1373176844790578, "clip_ratio/region_mean": 0.15371112618595362, "reward_total_mean": 0.6535242795944214, "reward_meter_mean": 0.6855431199073792, "reward_meter_std": 0.38864585757255554, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9865488409996033, "reward_repeat_soft_std": 0.01929342746734619, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.6535242795944214, "reward_total_composite_std": 0.22047758102416992} {"timestamp_utc": "2026-04-13T02:24:18Z", "mode": "train", "global_step": 1635, "epoch": 0.16423907584128578, "loss": -0.0864, "grad_norm": 4.243004322052002, "learning_rate": 5.048484848484849e-06, "num_tokens": 2949063.0, "completions/mean_length": 108.875, "completions/min_length": 50.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 51.28571701049805, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.9156553745269775, "rewards/meter/std": 0.1192454844713211, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.8868319392204285, "rewards/repeat_soft/std": 0.07442636787891388, "rewards/judge_quality/mean": 0.6349999904632568, "rewards/judge_quality/std": 0.33161941170692444, "rewards/total_composite/mean": 0.6561222672462463, "rewards/total_composite/std": 0.4120751917362213, "reward": 0.6561222672462463, "reward_std": 0.4120751917362213, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1171223372220993, "sampling/sampling_logp_difference/max": 1.46525239944458, "sampling/importance_sampling_ratio/min": 0.23101969063282013, "sampling/importance_sampling_ratio/mean": 1.0198975801467896, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6963787525892258, "clip_ratio/low_mean": 0.004999999888241291, "clip_ratio/low_min": 0.004999999888241291, "clip_ratio/high_mean": 0.08246122905984521, "clip_ratio/high_max": 0.08246122905984521, "clip_ratio/region_mean": 0.0874612289480865, "reward_total_mean": 0.6561222672462463, "reward_meter_mean": 0.9156553745269775, "reward_meter_std": 0.1192454844713211, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.8868319392204285, "reward_repeat_soft_std": 0.07442636787891388, "reward_judge_quality_mean": 0.6349999904632568, "reward_judge_quality_std": 0.33161941170692444, "reward_total_composite_mean": 0.6561222672462463, "reward_total_composite_std": 0.4120751917362213} {"timestamp_utc": "2026-04-13T02:24:26Z", "mode": "train", "global_step": 1636, "epoch": 0.16433952787543948, "loss": -0.0037, "grad_norm": 14.595754623413086, "learning_rate": 5.045454545454546e-06, "num_tokens": 2950637.0, "completions/mean_length": 44.75, "completions/min_length": 23.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.75, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.8803464770317078, "rewards/meter/std": 0.2609928250312805, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9812300205230713, "rewards/repeat_soft/std": 0.014209470711648464, "rewards/judge_quality/mean": 0.643750011920929, "rewards/judge_quality/std": 0.29031693935394287, "rewards/total_composite/mean": 0.81865394115448, "rewards/total_composite/std": 0.11597485095262527, "reward": 0.81865394115448, "reward_std": 0.11597482860088348, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14337892830371857, "sampling/sampling_logp_difference/max": 1.5138683319091797, "sampling/importance_sampling_ratio/min": 0.22005708515644073, "sampling/importance_sampling_ratio/mean": 1.005146861076355, "sampling/importance_sampling_ratio/max": 1.740054965019226, "entropy": 1.0020253956317902, "clip_ratio/low_mean": 0.04531904170289636, "clip_ratio/low_min": 0.04531904170289636, "clip_ratio/high_mean": 0.07672239188104868, "clip_ratio/high_max": 0.07672239188104868, "clip_ratio/region_mean": 0.12204143358394504, "reward_total_mean": 0.81865394115448, "reward_meter_mean": 0.8803464770317078, "reward_meter_std": 0.2609928250312805, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9812300205230713, "reward_repeat_soft_std": 0.014209470711648464, "reward_judge_quality_mean": 0.643750011920929, "reward_judge_quality_std": 0.29031693935394287, "reward_total_composite_mean": 0.81865394115448, "reward_total_composite_std": 0.11597485095262527} {"timestamp_utc": "2026-04-13T02:24:33Z", "mode": "train", "global_step": 1637, "epoch": 0.16443997990959316, "loss": 0.0249, "grad_norm": 10.971785545349121, "learning_rate": 5.042424242424243e-06, "num_tokens": 2952271.0, "completions/mean_length": 52.25, "completions/min_length": 45.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.25, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.8505172729492188, "rewards/meter/std": 0.26101648807525635, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9816761016845703, "rewards/repeat_soft/std": 0.01758272387087345, "rewards/judge_quality/mean": 0.5149999856948853, "rewards/judge_quality/std": 0.1810288280248642, "rewards/total_composite/mean": 0.785400390625, "rewards/total_composite/std": 0.08681700378656387, "reward": 0.785400390625, "reward_std": 0.08681700378656387, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13893598318099976, "sampling/sampling_logp_difference/max": 1.1338191032409668, "sampling/importance_sampling_ratio/min": 0.32180190086364746, "sampling/importance_sampling_ratio/mean": 1.0205283164978027, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.078762635588646, "clip_ratio/low_mean": 0.027308838441967964, "clip_ratio/low_min": 0.027308838441967964, "clip_ratio/high_mean": 0.08175899926573038, "clip_ratio/high_max": 0.08175899926573038, "clip_ratio/region_mean": 0.10906783770769835, "reward_total_mean": 0.785400390625, "reward_meter_mean": 0.8505172729492188, "reward_meter_std": 0.26101648807525635, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9816761016845703, "reward_repeat_soft_std": 0.01758272387087345, "reward_judge_quality_mean": 0.5149999856948853, "reward_judge_quality_std": 0.1810288280248642, "reward_total_composite_mean": 0.785400390625, "reward_total_composite_std": 0.08681700378656387} {"timestamp_utc": "2026-04-13T02:24:41Z", "mode": "train", "global_step": 1638, "epoch": 0.16454043194374687, "loss": -0.2148, "grad_norm": 4.936094284057617, "learning_rate": 5.0393939393939395e-06, "num_tokens": 2955137.0, "completions/mean_length": 154.25, "completions/min_length": 28.0, "completions/max_length": 197.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 154.25, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 197.0, "rewards/meter/mean": 0.9634156823158264, "rewards/meter/std": 0.06066402792930603, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.2828427255153656, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9297885298728943, "rewards/repeat_soft/std": 0.045450225472450256, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.7923909425735474, "rewards/total_composite/std": 0.07607455551624298, "reward": 0.7923909425735474, "reward_std": 0.07607456296682358, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14208850264549255, "sampling/sampling_logp_difference/max": 2.406545639038086, "sampling/importance_sampling_ratio/min": 0.09012608230113983, "sampling/importance_sampling_ratio/mean": 1.0158445835113525, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0841492041945457, "clip_ratio/low_mean": 0.03348214365541935, "clip_ratio/low_min": 0.03348214365541935, "clip_ratio/high_mean": 0.10172922350466251, "clip_ratio/high_max": 0.10172922350466251, "clip_ratio/region_mean": 0.13521136716008186, "reward_total_mean": 0.7923909425735474, "reward_meter_mean": 0.9634156823158264, "reward_meter_std": 0.06066402792930603, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.2828427255153656, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9297885298728943, "reward_repeat_soft_std": 0.045450225472450256, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.7923909425735474, "reward_total_composite_std": 0.07607455551624298} {"timestamp_utc": "2026-04-13T02:24:50Z", "mode": "train", "global_step": 1639, "epoch": 0.16464088397790055, "loss": 0.0158, "grad_norm": 11.660276412963867, "learning_rate": 5.036363636363637e-06, "num_tokens": 2956888.0, "completions/mean_length": 56.875, "completions/min_length": 50.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.875, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.8633304238319397, "rewards/meter/std": 0.2660471498966217, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9895181655883789, "rewards/repeat_soft/std": 0.015969563275575638, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.8009505271911621, "rewards/total_composite/std": 0.09565414488315582, "reward": 0.8009505271911621, "reward_std": 0.09565413743257523, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1630137860774994, "sampling/sampling_logp_difference/max": 1.6287202835083008, "sampling/importance_sampling_ratio/min": 0.19618046283721924, "sampling/importance_sampling_ratio/mean": 1.0090678930282593, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1675060838460922, "clip_ratio/low_mean": 0.048411883413791656, "clip_ratio/low_min": 0.048411883413791656, "clip_ratio/high_mean": 0.10953451786190271, "clip_ratio/high_max": 0.10953451786190271, "clip_ratio/region_mean": 0.15794640127569437, "reward_total_mean": 0.8009505271911621, "reward_meter_mean": 0.8633304238319397, "reward_meter_std": 0.2660471498966217, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9895181655883789, "reward_repeat_soft_std": 0.015969563275575638, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.8009505271911621, "reward_total_composite_std": 0.09565414488315582} {"timestamp_utc": "2026-04-13T02:24:57Z", "mode": "train", "global_step": 1640, "epoch": 0.16474133601205423, "loss": 0.0045, "grad_norm": 18.741300582885742, "learning_rate": 5.033333333333333e-06, "num_tokens": 2958526.0, "completions/mean_length": 37.75, "completions/min_length": 36.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.75, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.7991490364074707, "rewards/meter/std": 0.21541425585746765, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9834511280059814, "rewards/repeat_soft/std": 0.016943272203207016, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.7309621572494507, "rewards/total_composite/std": 0.09142913669347763, "reward": 0.7309621572494507, "reward_std": 0.09142914414405823, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09731552004814148, "sampling/sampling_logp_difference/max": 1.0089662075042725, "sampling/importance_sampling_ratio/min": 0.3645957112312317, "sampling/importance_sampling_ratio/mean": 1.0149294137954712, "sampling/importance_sampling_ratio/max": 1.8736473321914673, "entropy": 0.6369318105280399, "clip_ratio/low_mean": 0.04054053965955973, "clip_ratio/low_min": 0.04054053965955973, "clip_ratio/high_mean": 0.05520080588757992, "clip_ratio/high_max": 0.05520080588757992, "clip_ratio/region_mean": 0.09574134554713964, "reward_total_mean": 0.7309621572494507, "reward_meter_mean": 0.7991490364074707, "reward_meter_std": 0.21541425585746765, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9834511280059814, "reward_repeat_soft_std": 0.016943272203207016, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.7309621572494507, "reward_total_composite_std": 0.09142913669347763} {"timestamp_utc": "2026-04-13T02:25:04Z", "mode": "train", "global_step": 1641, "epoch": 0.16484178804620794, "loss": 0.0689, "grad_norm": 8.111538887023926, "learning_rate": 5.030303030303031e-06, "num_tokens": 2960547.0, "completions/mean_length": 83.625, "completions/min_length": 76.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.625, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9739391803741455, "rewards/meter/std": 0.0270667914301157, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.939361572265625, "rewards/repeat_soft/std": 0.08659057319164276, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.13265827298164368, "rewards/total_composite/mean": 0.8340837955474854, "rewards/total_composite/std": 0.047564007341861725, "reward": 0.8340837955474854, "reward_std": 0.04756401106715202, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12659481167793274, "sampling/sampling_logp_difference/max": 1.8322458267211914, "sampling/importance_sampling_ratio/min": 0.16005371510982513, "sampling/importance_sampling_ratio/mean": 1.013318419456482, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9364458620548248, "clip_ratio/low_mean": 0.08393598161637783, "clip_ratio/low_min": 0.08393598161637783, "clip_ratio/high_mean": 0.03125, "clip_ratio/high_max": 0.03125, "clip_ratio/region_mean": 0.11518598161637783, "reward_total_mean": 0.8340837955474854, "reward_meter_mean": 0.9739391803741455, "reward_meter_std": 0.0270667914301157, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.939361572265625, "reward_repeat_soft_std": 0.08659057319164276, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.13265827298164368, "reward_total_composite_mean": 0.8340837955474854, "reward_total_composite_std": 0.047564007341861725} {"timestamp_utc": "2026-04-13T02:25:11Z", "mode": "train", "global_step": 1642, "epoch": 0.16494224008036162, "loss": 0.0282, "grad_norm": 8.908906936645508, "learning_rate": 5.027272727272728e-06, "num_tokens": 2962987.0, "completions/mean_length": 104.0, "completions/min_length": 97.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.0, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.7357257604598999, "rewards/meter/std": 0.33823755383491516, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9197684526443481, "rewards/repeat_soft/std": 0.048282574862241745, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.6799284219741821, "rewards/total_composite/std": 0.14678692817687988, "reward": 0.6799284219741821, "reward_std": 0.14678692817687988, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11280792206525803, "sampling/sampling_logp_difference/max": 1.49139404296875, "sampling/importance_sampling_ratio/min": 0.2250586897134781, "sampling/importance_sampling_ratio/mean": 1.014775037765503, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6469286568462849, "clip_ratio/low_mean": 0.049665587954223156, "clip_ratio/low_min": 0.049665587954223156, "clip_ratio/high_mean": 0.05181799968704581, "clip_ratio/high_max": 0.05181799968704581, "clip_ratio/region_mean": 0.10148358764126897, "reward_total_mean": 0.6799284219741821, "reward_meter_mean": 0.7357257604598999, "reward_meter_std": 0.33823755383491516, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9197684526443481, "reward_repeat_soft_std": 0.048282574862241745, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.6799284219741821, "reward_total_composite_std": 0.14678692817687988} {"timestamp_utc": "2026-04-13T02:25:18Z", "mode": "train", "global_step": 1643, "epoch": 0.16504269211451533, "loss": 0.055, "grad_norm": 13.813337326049805, "learning_rate": 5.024242424242425e-06, "num_tokens": 2964551.0, "completions/mean_length": 44.5, "completions/min_length": 36.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9073149561882019, "rewards/meter/std": 0.09011561423540115, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9182060956954956, "rewards/repeat_soft/std": 0.05359184369444847, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7761123180389404, "rewards/total_composite/std": 0.04332376644015312, "reward": 0.7761123180389404, "reward_std": 0.04332374781370163, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11078031361103058, "sampling/sampling_logp_difference/max": 1.800649881362915, "sampling/importance_sampling_ratio/min": 0.2510451674461365, "sampling/importance_sampling_ratio/mean": 1.0171846151351929, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6164618916809559, "clip_ratio/low_mean": 0.03802083432674408, "clip_ratio/low_min": 0.03802083432674408, "clip_ratio/high_mean": 0.05423920135945082, "clip_ratio/high_max": 0.05423920135945082, "clip_ratio/region_mean": 0.0922600356861949, "reward_total_mean": 0.7761123180389404, "reward_meter_mean": 0.9073149561882019, "reward_meter_std": 0.09011561423540115, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9182060956954956, "reward_repeat_soft_std": 0.05359184369444847, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7761123180389404, "reward_total_composite_std": 0.04332376644015312} {"timestamp_utc": "2026-04-13T02:25:25Z", "mode": "train", "global_step": 1644, "epoch": 0.165143144148669, "loss": 0.0518, "grad_norm": 11.945354461669922, "learning_rate": 5.021212121212121e-06, "num_tokens": 2966218.0, "completions/mean_length": 48.375, "completions/min_length": 42.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.375, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.5527041554450989, "rewards/meter/std": 0.4083613157272339, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9827919006347656, "rewards/repeat_soft/std": 0.018428964540362358, "rewards/judge_quality/mean": 0.48000001907348633, "rewards/judge_quality/std": 0.19071295857429504, "rewards/total_composite/mean": 0.6409960389137268, "rewards/total_composite/std": 0.21496586501598358, "reward": 0.6409960389137268, "reward_std": 0.2149658501148224, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12490526586771011, "sampling/sampling_logp_difference/max": 1.299761414527893, "sampling/importance_sampling_ratio/min": 0.2725968062877655, "sampling/importance_sampling_ratio/mean": 1.0151194334030151, "sampling/importance_sampling_ratio/max": 1.7042194604873657, "entropy": 0.8850100710988045, "clip_ratio/low_mean": 0.06373292114585638, "clip_ratio/low_min": 0.06373292114585638, "clip_ratio/high_mean": 0.0551360547542572, "clip_ratio/high_max": 0.0551360547542572, "clip_ratio/region_mean": 0.11886897590011358, "reward_total_mean": 0.6409960389137268, "reward_meter_mean": 0.5527041554450989, "reward_meter_std": 0.4083613157272339, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9827919006347656, "reward_repeat_soft_std": 0.018428964540362358, "reward_judge_quality_mean": 0.48000001907348633, "reward_judge_quality_std": 0.19071295857429504, "reward_total_composite_mean": 0.6409960389137268, "reward_total_composite_std": 0.21496586501598358} {"timestamp_utc": "2026-04-13T02:25:38Z", "mode": "train", "global_step": 1645, "epoch": 0.1652435961828227, "loss": -0.0694, "grad_norm": 1.6027154922485352, "learning_rate": 5.0181818181818186e-06, "num_tokens": 2967638.0, "completions/mean_length": 151.5, "completions/min_length": 26.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 31.33333396911621, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.38199660181999207, "rewards/meter/std": 0.42244774103164673, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.518750011920929, "rewards/judge_quality/std": 0.3677513897418976, "rewards/total_composite/mean": 0.508429765701294, "rewards/total_composite/std": 0.3453904390335083, "reward": 0.508429765701294, "reward_std": 0.3453904390335083, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17765235900878906, "sampling/sampling_logp_difference/max": 2.183108329772949, "sampling/importance_sampling_ratio/min": 0.11269070208072662, "sampling/importance_sampling_ratio/mean": 1.0345908403396606, "sampling/importance_sampling_ratio/max": 1.9814064502716064, "entropy": 1.1580279469490051, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.1175394942983985, "clip_ratio/high_max": 0.1175394942983985, "clip_ratio/region_mean": 0.1175394942983985, "reward_total_mean": 0.508429765701294, "reward_meter_mean": 0.38199660181999207, "reward_meter_std": 0.42244774103164673, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.518750011920929, "reward_judge_quality_std": 0.3677513897418976, "reward_total_composite_mean": 0.508429765701294, "reward_total_composite_std": 0.3453904390335083} {"timestamp_utc": "2026-04-13T02:25:49Z", "mode": "train", "global_step": 1646, "epoch": 0.1653440482169764, "loss": -0.1016, "grad_norm": 2.9379446506500244, "learning_rate": 5.015151515151515e-06, "num_tokens": 2969168.0, "completions/mean_length": 98.25, "completions/min_length": 30.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 39.142860412597656, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.7638733386993408, "rewards/meter/std": 0.34906578063964844, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9752527475357056, "rewards/repeat_soft/std": 0.03836236521601677, "rewards/judge_quality/mean": 0.4437500238418579, "rewards/judge_quality/std": 0.23427319526672363, "rewards/total_composite/mean": 0.7150182723999023, "rewards/total_composite/std": 0.23654481768608093, "reward": 0.7150182723999023, "reward_std": 0.23654480278491974, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1367628276348114, "sampling/sampling_logp_difference/max": 2.4170949459075928, "sampling/importance_sampling_ratio/min": 0.08918032050132751, "sampling/importance_sampling_ratio/mean": 1.0340567827224731, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5384543091058731, "clip_ratio/low_mean": 0.01785714365541935, "clip_ratio/low_min": 0.01785714365541935, "clip_ratio/high_mean": 0.09563049208372831, "clip_ratio/high_max": 0.09563049208372831, "clip_ratio/region_mean": 0.11348763573914766, "reward_total_mean": 0.7150182723999023, "reward_meter_mean": 0.7638733386993408, "reward_meter_std": 0.34906578063964844, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9752527475357056, "reward_repeat_soft_std": 0.03836236521601677, "reward_judge_quality_mean": 0.4437500238418579, "reward_judge_quality_std": 0.23427319526672363, "reward_total_composite_mean": 0.7150182723999023, "reward_total_composite_std": 0.23654481768608093} {"timestamp_utc": "2026-04-13T02:25:56Z", "mode": "train", "global_step": 1647, "epoch": 0.16544450025113008, "loss": 0.0619, "grad_norm": 10.050203323364258, "learning_rate": 5.012121212121212e-06, "num_tokens": 2970804.0, "completions/mean_length": 56.5, "completions/min_length": 47.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.5, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.5762407779693604, "rewards/meter/std": 0.3693235516548157, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9941748380661011, "rewards/repeat_soft/std": 0.00809217244386673, "rewards/judge_quality/mean": 0.5862500071525574, "rewards/judge_quality/std": 0.2822834253311157, "rewards/total_composite/mean": 0.684600830078125, "rewards/total_composite/std": 0.19908873736858368, "reward": 0.684600830078125, "reward_std": 0.19908872246742249, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14762260019779205, "sampling/sampling_logp_difference/max": 1.6059250831604004, "sampling/importance_sampling_ratio/min": 0.20070379972457886, "sampling/importance_sampling_ratio/mean": 1.0183418989181519, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1184001043438911, "clip_ratio/low_mean": 0.05342600028961897, "clip_ratio/low_min": 0.05342600028961897, "clip_ratio/high_mean": 0.06690252758562565, "clip_ratio/high_max": 0.06690252758562565, "clip_ratio/region_mean": 0.12032852787524462, "reward_total_mean": 0.684600830078125, "reward_meter_mean": 0.5762407779693604, "reward_meter_std": 0.3693235516548157, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9941748380661011, "reward_repeat_soft_std": 0.00809217244386673, "reward_judge_quality_mean": 0.5862500071525574, "reward_judge_quality_std": 0.2822834253311157, "reward_total_composite_mean": 0.684600830078125, "reward_total_composite_std": 0.19908873736858368} {"timestamp_utc": "2026-04-13T02:26:03Z", "mode": "train", "global_step": 1648, "epoch": 0.16554495228528376, "loss": -0.047, "grad_norm": 11.253080368041992, "learning_rate": 5.009090909090909e-06, "num_tokens": 2972633.0, "completions/mean_length": 61.625, "completions/min_length": 54.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.625, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9892075061798096, "rewards/meter/std": 0.007771897129714489, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9514126181602478, "rewards/repeat_soft/std": 0.0655951276421547, "rewards/judge_quality/mean": 0.36000001430511475, "rewards/judge_quality/std": 0.09165150672197342, "rewards/total_composite/mean": 0.798284649848938, "rewards/total_composite/std": 0.02559744007885456, "reward": 0.798284649848938, "reward_std": 0.02559741772711277, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15665502846240997, "sampling/sampling_logp_difference/max": 2.7074475288391113, "sampling/importance_sampling_ratio/min": 0.06670685112476349, "sampling/importance_sampling_ratio/mean": 1.0174990892410278, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0336181670427322, "clip_ratio/low_mean": 0.035709102638065815, "clip_ratio/low_min": 0.035709102638065815, "clip_ratio/high_mean": 0.07832700666040182, "clip_ratio/high_max": 0.07832700666040182, "clip_ratio/region_mean": 0.11403610929846764, "reward_total_mean": 0.798284649848938, "reward_meter_mean": 0.9892075061798096, "reward_meter_std": 0.007771897129714489, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9514126181602478, "reward_repeat_soft_std": 0.0655951276421547, "reward_judge_quality_mean": 0.36000001430511475, "reward_judge_quality_std": 0.09165150672197342, "reward_total_composite_mean": 0.798284649848938, "reward_total_composite_std": 0.02559744007885456} {"timestamp_utc": "2026-04-13T02:26:10Z", "mode": "train", "global_step": 1649, "epoch": 0.16564540431943747, "loss": -0.028, "grad_norm": 11.557343482971191, "learning_rate": 5.006060606060607e-06, "num_tokens": 2974321.0, "completions/mean_length": 57.0, "completions/min_length": 50.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.0, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.8918443918228149, "rewards/meter/std": 0.24096351861953735, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9799614548683167, "rewards/repeat_soft/std": 0.021984798833727837, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.775326132774353, "rewards/total_composite/std": 0.10861766338348389, "reward": 0.775326132774353, "reward_std": 0.10861767083406448, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11690361052751541, "sampling/sampling_logp_difference/max": 1.3202157020568848, "sampling/importance_sampling_ratio/min": 0.2670776844024658, "sampling/importance_sampling_ratio/mean": 1.0183902978897095, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8249119743704796, "clip_ratio/low_mean": 0.017500000074505806, "clip_ratio/low_min": 0.017500000074505806, "clip_ratio/high_mean": 0.12666791398078203, "clip_ratio/high_max": 0.12666791398078203, "clip_ratio/region_mean": 0.14416791405528784, "reward_total_mean": 0.775326132774353, "reward_meter_mean": 0.8918443918228149, "reward_meter_std": 0.24096351861953735, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9799614548683167, "reward_repeat_soft_std": 0.021984798833727837, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.775326132774353, "reward_total_composite_std": 0.10861766338348389} {"timestamp_utc": "2026-04-13T02:26:17Z", "mode": "train", "global_step": 1650, "epoch": 0.16574585635359115, "loss": -0.0185, "grad_norm": 11.835944175720215, "learning_rate": 5.003030303030303e-06, "num_tokens": 2976310.0, "completions/mean_length": 66.625, "completions/min_length": 63.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.625, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9800817966461182, "rewards/meter/std": 0.03898761048913002, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9698901176452637, "rewards/repeat_soft/std": 0.035950854420661926, "rewards/judge_quality/mean": 0.4437499940395355, "rewards/judge_quality/std": 0.128834068775177, "rewards/total_composite/mean": 0.8211508393287659, "rewards/total_composite/std": 0.04620610922574997, "reward": 0.8211508393287659, "reward_std": 0.04620611295104027, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14691869914531708, "sampling/sampling_logp_difference/max": 1.5870885848999023, "sampling/importance_sampling_ratio/min": 0.20452018082141876, "sampling/importance_sampling_ratio/mean": 1.0193060636520386, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.223557859659195, "clip_ratio/low_mean": 0.062119645066559315, "clip_ratio/low_min": 0.062119645066559315, "clip_ratio/high_mean": 0.07799414359033108, "clip_ratio/high_max": 0.07799414359033108, "clip_ratio/region_mean": 0.1401137886568904, "reward_total_mean": 0.8211508393287659, "reward_meter_mean": 0.9800817966461182, "reward_meter_std": 0.03898761048913002, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9698901176452637, "reward_repeat_soft_std": 0.035950854420661926, "reward_judge_quality_mean": 0.4437499940395355, "reward_judge_quality_std": 0.128834068775177, "reward_total_composite_mean": 0.8211508393287659, "reward_total_composite_std": 0.04620610922574997} {"timestamp_utc": "2026-04-13T02:27:29Z", "mode": "eval", "global_step": 1650, "epoch": 0.16574585635359115, "eval_loss": NaN, "eval_runtime": 72.2411, "eval_samples_per_second": 1.107, "eval_steps_per_second": 0.138, "eval_num_tokens": 2976310.0, "eval_completions/mean_length": 109.1625, "eval_completions/min_length": 36.9, "eval_completions/max_length": 268.9, "eval_completions/clipped_ratio": 0.05, "eval_completions/mean_terminated_length": 87.03154907226562, "eval_completions/min_terminated_length": 36.9, "eval_completions/max_terminated_length": 150.7, "eval_rewards/meter/mean": 0.7963314950466156, "eval_rewards/meter/std": 0.26094688987359405, "eval_rewards/count_adherence/mean": 0.9833333313465118, "eval_rewards/count_adherence/std": 0.04714045412838459, "eval_rewards/hard_gate/mean": 0.95, "eval_rewards/hard_gate/std": 0.11700168251991272, "eval_rewards/repeat_soft/mean": 0.9324155032634736, "eval_rewards/repeat_soft/std": 0.08431173684075474, "eval_rewards/judge_quality/mean": 0.40999999046325686, "eval_rewards/judge_quality/std": 0.09943610830232501, "eval_rewards/total_composite/mean": 0.7030746102333069, "eval_rewards/total_composite/std": 0.16358539275825024, "eval_reward": 0.7030746102333069, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.07194178961217404, "eval_sampling/sampling_logp_difference/max": 1.0455205917358399, "eval_sampling/importance_sampling_ratio/min": 0.3586724102497101, "eval_sampling/importance_sampling_ratio/mean": 1.0168430924415588, "eval_sampling/importance_sampling_ratio/max": 1.4330664157867432, "eval_entropy": 0.7998869359493256, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7030746102333069, "eval_reward_meter_mean": 0.7963314950466156, "eval_reward_meter_std": 0.26094688987359405, "eval_reward_count_adherence_mean": 0.9833333313465118, "eval_reward_count_adherence_std": 0.04714045412838459, "eval_reward_hard_gate_mean": 0.95, "eval_reward_hard_gate_std": 0.11700168251991272, "eval_reward_repeat_soft_mean": 0.9324155032634736, "eval_reward_repeat_soft_std": 0.08431173684075474, "eval_reward_judge_quality_mean": 0.40999999046325686, "eval_reward_judge_quality_std": 0.09943610830232501, "eval_reward_total_composite_mean": 0.7030746102333069, "eval_reward_total_composite_std": 0.16358539275825024} {"timestamp_utc": "2026-04-13T02:27:39Z", "mode": "train", "global_step": 1651, "epoch": 0.16584630838774486, "loss": 0.007, "grad_norm": 12.213820457458496, "learning_rate": 5e-06, "num_tokens": 2977895.0, "completions/mean_length": 56.125, "completions/min_length": 47.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.125, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.6616643667221069, "rewards/meter/std": 0.2913151681423187, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9913097620010376, "rewards/repeat_soft/std": 0.008262117393314838, "rewards/judge_quality/mean": 0.4649999737739563, "rewards/judge_quality/std": 0.14322809875011444, "rewards/total_composite/mean": 0.6863799095153809, "rewards/total_composite/std": 0.1418912261724472, "reward": 0.6863799095153809, "reward_std": 0.1418912261724472, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15563564002513885, "sampling/sampling_logp_difference/max": 1.650242805480957, "sampling/importance_sampling_ratio/min": 0.1920032948255539, "sampling/importance_sampling_ratio/mean": 1.0144966840744019, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3180860877037048, "clip_ratio/low_mean": 0.05438305251300335, "clip_ratio/low_min": 0.05438305251300335, "clip_ratio/high_mean": 0.08413161616772413, "clip_ratio/high_max": 0.08413161616772413, "clip_ratio/region_mean": 0.13851466868072748, "reward_total_mean": 0.6863799095153809, "reward_meter_mean": 0.6616643667221069, "reward_meter_std": 0.2913151681423187, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9913097620010376, "reward_repeat_soft_std": 0.008262117393314838, "reward_judge_quality_mean": 0.4649999737739563, "reward_judge_quality_std": 0.14322809875011444, "reward_total_composite_mean": 0.6863799095153809, "reward_total_composite_std": 0.1418912261724472} {"timestamp_utc": "2026-04-13T02:27:46Z", "mode": "train", "global_step": 1652, "epoch": 0.16594676042189854, "loss": 0.0139, "grad_norm": 8.242161750793457, "learning_rate": 4.996969696969698e-06, "num_tokens": 2979836.0, "completions/mean_length": 71.625, "completions/min_length": 66.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.625, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.9801841378211975, "rewards/meter/std": 0.013890571892261505, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9823721647262573, "rewards/repeat_soft/std": 0.013974052853882313, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.8201950788497925, "rewards/total_composite/std": 0.04200548306107521, "reward": 0.8201950788497925, "reward_std": 0.04200548306107521, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11437610536813736, "sampling/sampling_logp_difference/max": 1.1927334070205688, "sampling/importance_sampling_ratio/min": 0.35816219449043274, "sampling/importance_sampling_ratio/mean": 1.0159757137298584, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.83498465269804, "clip_ratio/low_mean": 0.0826195776462555, "clip_ratio/low_min": 0.0826195776462555, "clip_ratio/high_mean": 0.042857143096625805, "clip_ratio/high_max": 0.042857143096625805, "clip_ratio/region_mean": 0.1254767207428813, "reward_total_mean": 0.8201950788497925, "reward_meter_mean": 0.9801841378211975, "reward_meter_std": 0.013890571892261505, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9823721647262573, "reward_repeat_soft_std": 0.013974052853882313, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.8201950788497925, "reward_total_composite_std": 0.04200548306107521} {"timestamp_utc": "2026-04-13T02:27:54Z", "mode": "train", "global_step": 1653, "epoch": 0.16604721245605222, "loss": -0.0055, "grad_norm": 6.507480621337891, "learning_rate": 4.993939393939394e-06, "num_tokens": 2982130.0, "completions/mean_length": 116.75, "completions/min_length": 108.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.75, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.9804592728614807, "rewards/meter/std": 0.019005445763468742, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8363834619522095, "rewards/repeat_soft/std": 0.12799866497516632, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.6825127005577087, "rewards/total_composite/std": 0.27692103385925293, "reward": 0.6825127005577087, "reward_std": 0.27692103385925293, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11734841763973236, "sampling/sampling_logp_difference/max": 1.7580832242965698, "sampling/importance_sampling_ratio/min": 0.17237494885921478, "sampling/importance_sampling_ratio/mean": 1.029596209526062, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8375021144747734, "clip_ratio/low_mean": 0.0173611119389534, "clip_ratio/low_min": 0.0173611119389534, "clip_ratio/high_mean": 0.08196329278871417, "clip_ratio/high_max": 0.08196329278871417, "clip_ratio/region_mean": 0.09932440472766757, "reward_total_mean": 0.6825127005577087, "reward_meter_mean": 0.9804592728614807, "reward_meter_std": 0.019005445763468742, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8363834619522095, "reward_repeat_soft_std": 0.12799866497516632, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.6825127005577087, "reward_total_composite_std": 0.27692103385925293} {"timestamp_utc": "2026-04-13T02:28:01Z", "mode": "train", "global_step": 1654, "epoch": 0.16614766449020593, "loss": 0.0391, "grad_norm": 12.08857250213623, "learning_rate": 4.990909090909091e-06, "num_tokens": 2983870.0, "completions/mean_length": 57.5, "completions/min_length": 51.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.5, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9688901901245117, "rewards/meter/std": 0.04815753549337387, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9283440113067627, "rewards/repeat_soft/std": 0.0940738171339035, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8070849776268005, "rewards/total_composite/std": 0.02311290055513382, "reward": 0.8070849776268005, "reward_std": 0.023112913593649864, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12370050698518753, "sampling/sampling_logp_difference/max": 1.1854352951049805, "sampling/importance_sampling_ratio/min": 0.30561313033103943, "sampling/importance_sampling_ratio/mean": 1.0066584348678589, "sampling/importance_sampling_ratio/max": 1.8779805898666382, "entropy": 0.922354519367218, "clip_ratio/low_mean": 0.050630198791623116, "clip_ratio/low_min": 0.050630198791623116, "clip_ratio/high_mean": 0.09820129815489054, "clip_ratio/high_max": 0.09820129815489054, "clip_ratio/region_mean": 0.14883149694651365, "reward_total_mean": 0.8070849776268005, "reward_meter_mean": 0.9688901901245117, "reward_meter_std": 0.04815753549337387, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9283440113067627, "reward_repeat_soft_std": 0.0940738171339035, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8070849776268005, "reward_total_composite_std": 0.02311290055513382} {"timestamp_utc": "2026-04-13T02:28:09Z", "mode": "train", "global_step": 1655, "epoch": 0.1662481165243596, "loss": -0.0363, "grad_norm": 9.825575828552246, "learning_rate": 4.987878787878789e-06, "num_tokens": 2985492.0, "completions/mean_length": 36.75, "completions/min_length": 33.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.6481068134307861, "rewards/meter/std": 0.39635545015335083, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9074042439460754, "rewards/repeat_soft/std": 0.10441696643829346, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.6617634892463684, "rewards/total_composite/std": 0.17213323712348938, "reward": 0.6617634892463684, "reward_std": 0.17213323712348938, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08039863407611847, "sampling/sampling_logp_difference/max": 1.1497935056686401, "sampling/importance_sampling_ratio/min": 0.31719574332237244, "sampling/importance_sampling_ratio/mean": 1.005466103553772, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.422570563852787, "clip_ratio/low_mean": 0.038590796291828156, "clip_ratio/low_min": 0.038590796291828156, "clip_ratio/high_mean": 0.04191169259138405, "clip_ratio/high_max": 0.04191169259138405, "clip_ratio/region_mean": 0.08050248888321221, "reward_total_mean": 0.6617634892463684, "reward_meter_mean": 0.6481068134307861, "reward_meter_std": 0.39635545015335083, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9074042439460754, "reward_repeat_soft_std": 0.10441696643829346, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.6617634892463684, "reward_total_composite_std": 0.17213323712348938} {"timestamp_utc": "2026-04-13T02:28:15Z", "mode": "train", "global_step": 1656, "epoch": 0.16634856855851332, "loss": 0.0034, "grad_norm": 27.29120445251465, "learning_rate": 4.984848484848485e-06, "num_tokens": 2987051.0, "completions/mean_length": 26.875, "completions/min_length": 24.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.875, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.8744431734085083, "rewards/meter/std": 0.2936907410621643, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.3050000071525574, "rewards/judge_quality/std": 0.14322808384895325, "rewards/total_composite/mean": 0.7312494516372681, "rewards/total_composite/std": 0.12216483056545258, "reward": 0.7312494516372681, "reward_std": 0.12216481566429138, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17205914855003357, "sampling/sampling_logp_difference/max": 1.1666207313537598, "sampling/importance_sampling_ratio/min": 0.31141752004623413, "sampling/importance_sampling_ratio/mean": 1.0346202850341797, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3321965336799622, "clip_ratio/low_mean": 0.0520833320915699, "clip_ratio/low_min": 0.0520833320915699, "clip_ratio/high_mean": 0.157102320343256, "clip_ratio/high_max": 0.157102320343256, "clip_ratio/region_mean": 0.2091856524348259, "reward_total_mean": 0.7312494516372681, "reward_meter_mean": 0.8744431734085083, "reward_meter_std": 0.2936907410621643, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.3050000071525574, "reward_judge_quality_std": 0.14322808384895325, "reward_total_composite_mean": 0.7312494516372681, "reward_total_composite_std": 0.12216483056545258} {"timestamp_utc": "2026-04-13T02:28:26Z", "mode": "train", "global_step": 1657, "epoch": 0.166449020592667, "loss": -0.1448, "grad_norm": 1.8161120414733887, "learning_rate": 4.981818181818182e-06, "num_tokens": 2988891.0, "completions/mean_length": 191.0, "completions/min_length": 82.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 84.0, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.8195592164993286, "rewards/meter/std": 0.2775106430053711, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9797635674476624, "rewards/repeat_soft/std": 0.02518884837627411, "rewards/judge_quality/mean": 0.3687500059604645, "rewards/judge_quality/std": 0.27559998631477356, "rewards/total_composite/mean": 0.6830203533172607, "rewards/total_composite/std": 0.29874566197395325, "reward": 0.6830203533172607, "reward_std": 0.29874566197395325, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13683032989501953, "sampling/sampling_logp_difference/max": 1.5010852813720703, "sampling/importance_sampling_ratio/min": 0.22288814187049866, "sampling/importance_sampling_ratio/mean": 0.9924822449684143, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7523489892482758, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10700379218906164, "clip_ratio/high_max": 0.10700379218906164, "clip_ratio/region_mean": 0.10700379218906164, "reward_total_mean": 0.6830203533172607, "reward_meter_mean": 0.8195592164993286, "reward_meter_std": 0.2775106430053711, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9797635674476624, "reward_repeat_soft_std": 0.02518884837627411, "reward_judge_quality_mean": 0.3687500059604645, "reward_judge_quality_std": 0.27559998631477356, "reward_total_composite_mean": 0.6830203533172607, "reward_total_composite_std": 0.29874566197395325} {"timestamp_utc": "2026-04-13T02:28:34Z", "mode": "train", "global_step": 1658, "epoch": 0.16654947262682068, "loss": 0.028, "grad_norm": 7.544069290161133, "learning_rate": 4.978787878787879e-06, "num_tokens": 2991097.0, "completions/mean_length": 94.75, "completions/min_length": 93.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.75, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.9782789349555969, "rewards/meter/std": 0.013471477665007114, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8578376770019531, "rewards/repeat_soft/std": 0.09590056538581848, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.8068842887878418, "rewards/total_composite/std": 0.044822514057159424, "reward": 0.8068842887878418, "reward_std": 0.04482252523303032, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12531179189682007, "sampling/sampling_logp_difference/max": 1.5102384090423584, "sampling/importance_sampling_ratio/min": 0.22085732221603394, "sampling/importance_sampling_ratio/mean": 1.005642294883728, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7051829844713211, "clip_ratio/low_mean": 0.04698191583156586, "clip_ratio/low_min": 0.04698191583156586, "clip_ratio/high_mean": 0.06278597563505173, "clip_ratio/high_max": 0.06278597563505173, "clip_ratio/region_mean": 0.10976789146661758, "reward_total_mean": 0.8068842887878418, "reward_meter_mean": 0.9782789349555969, "reward_meter_std": 0.013471477665007114, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8578376770019531, "reward_repeat_soft_std": 0.09590056538581848, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.8068842887878418, "reward_total_composite_std": 0.044822514057159424} {"timestamp_utc": "2026-04-13T02:28:42Z", "mode": "train", "global_step": 1659, "epoch": 0.1666499246609744, "loss": 0.0639, "grad_norm": 8.737678527832031, "learning_rate": 4.975757575757576e-06, "num_tokens": 2993541.0, "completions/mean_length": 111.5, "completions/min_length": 99.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.5, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.7854660153388977, "rewards/meter/std": 0.2790258228778839, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8820295333862305, "rewards/repeat_soft/std": 0.10696180164813995, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.7253502011299133, "rewards/total_composite/std": 0.14672935009002686, "reward": 0.7253502011299133, "reward_std": 0.14672936499118805, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1277361661195755, "sampling/sampling_logp_difference/max": 1.6165847778320312, "sampling/importance_sampling_ratio/min": 0.1985757201910019, "sampling/importance_sampling_ratio/mean": 1.01046884059906, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8805188424885273, "clip_ratio/low_mean": 0.040955882519483566, "clip_ratio/low_min": 0.040955882519483566, "clip_ratio/high_mean": 0.05376736354082823, "clip_ratio/high_max": 0.05376736354082823, "clip_ratio/region_mean": 0.0947232460603118, "reward_total_mean": 0.7253502011299133, "reward_meter_mean": 0.7854660153388977, "reward_meter_std": 0.2790258228778839, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8820295333862305, "reward_repeat_soft_std": 0.10696180164813995, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.7253502011299133, "reward_total_composite_std": 0.14672935009002686} {"timestamp_utc": "2026-04-13T02:28:48Z", "mode": "train", "global_step": 1660, "epoch": 0.16675037669512807, "loss": 0.0727, "grad_norm": 13.58033275604248, "learning_rate": 4.972727272727273e-06, "num_tokens": 2995238.0, "completions/mean_length": 47.125, "completions/min_length": 39.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.125, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9089680910110474, "rewards/meter/std": 0.23293237388134003, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9793822765350342, "rewards/repeat_soft/std": 0.025264045223593712, "rewards/judge_quality/mean": 0.476250022649765, "rewards/judge_quality/std": 0.19167962670326233, "rewards/total_composite/mean": 0.7998489141464233, "rewards/total_composite/std": 0.12153846770524979, "reward": 0.7998489141464233, "reward_std": 0.12153847515583038, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13172809779644012, "sampling/sampling_logp_difference/max": 1.3448967933654785, "sampling/importance_sampling_ratio/min": 0.2605665922164917, "sampling/importance_sampling_ratio/mean": 1.0024346113204956, "sampling/importance_sampling_ratio/max": 1.7715846300125122, "entropy": 0.8799888044595718, "clip_ratio/low_mean": 0.036793564446270466, "clip_ratio/low_min": 0.036793564446270466, "clip_ratio/high_mean": 0.09371008072048426, "clip_ratio/high_max": 0.09371008072048426, "clip_ratio/region_mean": 0.13050364516675472, "reward_total_mean": 0.7998489141464233, "reward_meter_mean": 0.9089680910110474, "reward_meter_std": 0.23293237388134003, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9793822765350342, "reward_repeat_soft_std": 0.025264045223593712, "reward_judge_quality_mean": 0.476250022649765, "reward_judge_quality_std": 0.19167962670326233, "reward_total_composite_mean": 0.7998489141464233, "reward_total_composite_std": 0.12153846770524979} {"timestamp_utc": "2026-04-13T02:28:55Z", "mode": "train", "global_step": 1661, "epoch": 0.16685082872928178, "loss": -0.0185, "grad_norm": 9.408683776855469, "learning_rate": 4.9696969696969696e-06, "num_tokens": 2997092.0, "completions/mean_length": 66.75, "completions/min_length": 60.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.75, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.8846759796142578, "rewards/meter/std": 0.2935939431190491, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9785652160644531, "rewards/repeat_soft/std": 0.01975269801914692, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7655857801437378, "rewards/total_composite/std": 0.1296885758638382, "reward": 0.7655857801437378, "reward_std": 0.129688560962677, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14988386631011963, "sampling/sampling_logp_difference/max": 1.8619866371154785, "sampling/importance_sampling_ratio/min": 0.15536366403102875, "sampling/importance_sampling_ratio/mean": 1.013256549835205, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.005303494632244, "clip_ratio/low_mean": 0.014583333395421505, "clip_ratio/low_min": 0.014583333395421505, "clip_ratio/high_mean": 0.14039989374578, "clip_ratio/high_max": 0.14039989374578, "clip_ratio/region_mean": 0.1549832271412015, "reward_total_mean": 0.7655857801437378, "reward_meter_mean": 0.8846759796142578, "reward_meter_std": 0.2935939431190491, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9785652160644531, "reward_repeat_soft_std": 0.01975269801914692, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7655857801437378, "reward_total_composite_std": 0.1296885758638382} {"timestamp_utc": "2026-04-13T02:29:03Z", "mode": "train", "global_step": 1662, "epoch": 0.16695128076343546, "loss": 0.0119, "grad_norm": 6.0860981941223145, "learning_rate": 4.966666666666667e-06, "num_tokens": 2999413.0, "completions/mean_length": 125.125, "completions/min_length": 116.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.125, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.8932930827140808, "rewards/meter/std": 0.11194364726543427, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7712917327880859, "rewards/repeat_soft/std": 0.18319423496723175, "rewards/judge_quality/mean": 0.3725000023841858, "rewards/judge_quality/std": 0.1947709321975708, "rewards/total_composite/mean": 0.6479024291038513, "rewards/total_composite/std": 0.2698705494403839, "reward": 0.6479024291038513, "reward_std": 0.2698705494403839, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11181729286909103, "sampling/sampling_logp_difference/max": 1.9692459106445312, "sampling/importance_sampling_ratio/min": 0.13956205546855927, "sampling/importance_sampling_ratio/mean": 1.0113366842269897, "sampling/importance_sampling_ratio/max": 1.9852772951126099, "entropy": 0.8461880646646023, "clip_ratio/low_mean": 0.023185361176729202, "clip_ratio/low_min": 0.023185361176729202, "clip_ratio/high_mean": 0.07389063318260014, "clip_ratio/high_max": 0.07389063318260014, "clip_ratio/region_mean": 0.09707599435932934, "reward_total_mean": 0.6479024291038513, "reward_meter_mean": 0.8932930827140808, "reward_meter_std": 0.11194364726543427, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7712917327880859, "reward_repeat_soft_std": 0.18319423496723175, "reward_judge_quality_mean": 0.3725000023841858, "reward_judge_quality_std": 0.1947709321975708, "reward_total_composite_mean": 0.6479024291038513, "reward_total_composite_std": 0.2698705494403839} {"timestamp_utc": "2026-04-13T02:29:15Z", "mode": "train", "global_step": 1663, "epoch": 0.16705173279758914, "loss": -0.1429, "grad_norm": 2.249655246734619, "learning_rate": 4.963636363636364e-06, "num_tokens": 3001146.0, "completions/mean_length": 112.625, "completions/min_length": 47.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 55.57143020629883, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9267905950546265, "rewards/meter/std": 0.13530771434307098, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9914414286613464, "rewards/repeat_soft/std": 0.014372862875461578, "rewards/judge_quality/mean": 0.48124998807907104, "rewards/judge_quality/std": 0.2531762421131134, "rewards/total_composite/mean": 0.7442009449005127, "rewards/total_composite/std": 0.30648958683013916, "reward": 0.7442009449005127, "reward_std": 0.3064895570278168, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16582725942134857, "sampling/sampling_logp_difference/max": 3.4340200424194336, "sampling/importance_sampling_ratio/min": 0.03225700557231903, "sampling/importance_sampling_ratio/mean": 1.0322494506835938, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0707557499408722, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.14168349653482437, "clip_ratio/high_max": 0.14168349653482437, "clip_ratio/region_mean": 0.14168349653482437, "reward_total_mean": 0.7442009449005127, "reward_meter_mean": 0.9267905950546265, "reward_meter_std": 0.13530771434307098, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9914414286613464, "reward_repeat_soft_std": 0.014372862875461578, "reward_judge_quality_mean": 0.48124998807907104, "reward_judge_quality_std": 0.2531762421131134, "reward_total_composite_mean": 0.7442009449005127, "reward_total_composite_std": 0.30648958683013916} {"timestamp_utc": "2026-04-13T02:29:23Z", "mode": "train", "global_step": 1664, "epoch": 0.16715218483174285, "loss": -0.015, "grad_norm": 13.280587196350098, "learning_rate": 4.9606060606060605e-06, "num_tokens": 3002506.0, "completions/mean_length": 32.0, "completions/min_length": 29.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9902047514915466, "rewards/meter/std": 0.0056441472843289375, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9554632902145386, "rewards/repeat_soft/std": 0.01990281604230404, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8238884210586548, "rewards/total_composite/std": 0.004088605288416147, "reward": 0.8238884210586548, "reward_std": 0.00408860994502902, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14705711603164673, "sampling/sampling_logp_difference/max": 1.001070261001587, "sampling/importance_sampling_ratio/min": 0.36805805563926697, "sampling/importance_sampling_ratio/mean": 1.0344033241271973, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.07106514275074, "clip_ratio/low_mean": 0.055228848941624165, "clip_ratio/low_min": 0.055228848941624165, "clip_ratio/high_mean": 0.08148297155275941, "clip_ratio/high_max": 0.08148297155275941, "clip_ratio/region_mean": 0.13671182049438357, "reward_total_mean": 0.8238884210586548, "reward_meter_mean": 0.9902047514915466, "reward_meter_std": 0.0056441472843289375, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9554632902145386, "reward_repeat_soft_std": 0.01990281604230404, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8238884210586548, "reward_total_composite_std": 0.004088605288416147} {"timestamp_utc": "2026-04-13T02:29:31Z", "mode": "train", "global_step": 1665, "epoch": 0.16725263686589653, "loss": 0.0396, "grad_norm": 13.470144271850586, "learning_rate": 4.957575757575758e-06, "num_tokens": 3004339.0, "completions/mean_length": 58.125, "completions/min_length": 48.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.125, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.7669980525970459, "rewards/meter/std": 0.26245108246803284, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9303954839706421, "rewards/repeat_soft/std": 0.11064966768026352, "rewards/judge_quality/mean": 0.3474999964237213, "rewards/judge_quality/std": 0.11310551315546036, "rewards/total_composite/mean": 0.6924386024475098, "rewards/total_composite/std": 0.10563117265701294, "reward": 0.6924386024475098, "reward_std": 0.10563115775585175, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11673540621995926, "sampling/sampling_logp_difference/max": 2.290830612182617, "sampling/importance_sampling_ratio/min": 0.10118238627910614, "sampling/importance_sampling_ratio/mean": 1.0189331769943237, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6529494524002075, "clip_ratio/low_mean": 0.04477653466165066, "clip_ratio/low_min": 0.04477653466165066, "clip_ratio/high_mean": 0.05984456650912762, "clip_ratio/high_max": 0.05984456650912762, "clip_ratio/region_mean": 0.10462110117077827, "reward_total_mean": 0.6924386024475098, "reward_meter_mean": 0.7669980525970459, "reward_meter_std": 0.26245108246803284, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9303954839706421, "reward_repeat_soft_std": 0.11064966768026352, "reward_judge_quality_mean": 0.3474999964237213, "reward_judge_quality_std": 0.11310551315546036, "reward_total_composite_mean": 0.6924386024475098, "reward_total_composite_std": 0.10563117265701294} {"timestamp_utc": "2026-04-13T02:29:38Z", "mode": "train", "global_step": 1666, "epoch": 0.16735308890005024, "loss": 0.0552, "grad_norm": 21.74496841430664, "learning_rate": 4.954545454545455e-06, "num_tokens": 3005779.0, "completions/mean_length": 32.0, "completions/min_length": 27.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.7792615294456482, "rewards/meter/std": 0.38021478056907654, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9432221055030823, "rewards/repeat_soft/std": 0.04442070797085762, "rewards/judge_quality/mean": 0.36249998211860657, "rewards/judge_quality/std": 0.12464234232902527, "rewards/total_composite/mean": 0.7037399411201477, "rewards/total_composite/std": 0.1938115656375885, "reward": 0.7037399411201477, "reward_std": 0.1938115507364273, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16075797379016876, "sampling/sampling_logp_difference/max": 1.0762553215026855, "sampling/importance_sampling_ratio/min": 0.34086957573890686, "sampling/importance_sampling_ratio/mean": 1.0149304866790771, "sampling/importance_sampling_ratio/max": 1.9735372066497803, "entropy": 1.1543890833854675, "clip_ratio/low_mean": 0.05539215914905071, "clip_ratio/low_min": 0.05539215914905071, "clip_ratio/high_mean": 0.10090024210512638, "clip_ratio/high_max": 0.10090024210512638, "clip_ratio/region_mean": 0.1562924012541771, "reward_total_mean": 0.7037399411201477, "reward_meter_mean": 0.7792615294456482, "reward_meter_std": 0.38021478056907654, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9432221055030823, "reward_repeat_soft_std": 0.04442070797085762, "reward_judge_quality_mean": 0.36249998211860657, "reward_judge_quality_std": 0.12464234232902527, "reward_total_composite_mean": 0.7037399411201477, "reward_total_composite_std": 0.1938115656375885} {"timestamp_utc": "2026-04-13T02:29:47Z", "mode": "train", "global_step": 1667, "epoch": 0.16745354093420392, "loss": 0.0183, "grad_norm": 7.239680767059326, "learning_rate": 4.951515151515152e-06, "num_tokens": 3008588.0, "completions/mean_length": 173.125, "completions/min_length": 156.0, "completions/max_length": 186.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 173.125, "completions/min_terminated_length": 156.0, "completions/max_terminated_length": 186.0, "rewards/meter/mean": 0.7835441827774048, "rewards/meter/std": 0.32737094163894653, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9307234883308411, "rewards/repeat_soft/std": 0.04844508692622185, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.6775422096252441, "rewards/total_composite/std": 0.15218612551689148, "reward": 0.6775422096252441, "reward_std": 0.15218614041805267, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12207566201686859, "sampling/sampling_logp_difference/max": 2.0101280212402344, "sampling/importance_sampling_ratio/min": 0.13397152721881866, "sampling/importance_sampling_ratio/mean": 1.014515995979309, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.724847212433815, "clip_ratio/low_mean": 0.02538580261170864, "clip_ratio/low_min": 0.02538580261170864, "clip_ratio/high_mean": 0.08833872526884079, "clip_ratio/high_max": 0.08833872526884079, "clip_ratio/region_mean": 0.11372452788054943, "reward_total_mean": 0.6775422096252441, "reward_meter_mean": 0.7835441827774048, "reward_meter_std": 0.32737094163894653, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9307234883308411, "reward_repeat_soft_std": 0.04844508692622185, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.6775422096252441, "reward_total_composite_std": 0.15218612551689148} {"timestamp_utc": "2026-04-13T02:29:56Z", "mode": "train", "global_step": 1668, "epoch": 0.1675539929683576, "loss": 0.035, "grad_norm": 6.882042407989502, "learning_rate": 4.9484848484848495e-06, "num_tokens": 3011172.0, "completions/mean_length": 130.0, "completions/min_length": 109.0, "completions/max_length": 139.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 130.0, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.5513296127319336, "rewards/meter/std": 0.27569150924682617, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9672142863273621, "rewards/repeat_soft/std": 0.013382477685809135, "rewards/judge_quality/mean": 0.24250000715255737, "rewards/judge_quality/std": 0.11792854964733124, "rewards/total_composite/mean": 0.5675697326660156, "rewards/total_composite/std": 0.13029181957244873, "reward": 0.5675697326660156, "reward_std": 0.13029180467128754, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12815827131271362, "sampling/sampling_logp_difference/max": 1.9973902702331543, "sampling/importance_sampling_ratio/min": 0.13568894565105438, "sampling/importance_sampling_ratio/mean": 1.0036228895187378, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7615366876125336, "clip_ratio/low_mean": 0.05510352831333876, "clip_ratio/low_min": 0.05510352831333876, "clip_ratio/high_mean": 0.06894950941205025, "clip_ratio/high_max": 0.06894950941205025, "clip_ratio/region_mean": 0.124053037725389, "reward_total_mean": 0.5675697326660156, "reward_meter_mean": 0.5513296127319336, "reward_meter_std": 0.27569150924682617, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9672142863273621, "reward_repeat_soft_std": 0.013382477685809135, "reward_judge_quality_mean": 0.24250000715255737, "reward_judge_quality_std": 0.11792854964733124, "reward_total_composite_mean": 0.5675697326660156, "reward_total_composite_std": 0.13029181957244873} {"timestamp_utc": "2026-04-13T02:30:03Z", "mode": "train", "global_step": 1669, "epoch": 0.1676544450025113, "loss": 0.0698, "grad_norm": 42.84245681762695, "learning_rate": 4.945454545454546e-06, "num_tokens": 3012600.0, "completions/mean_length": 31.5, "completions/min_length": 29.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.5, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.8724067211151123, "rewards/meter/std": 0.3415348529815674, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4124999940395355, "rewards/judge_quality/std": 0.1060660108923912, "rewards/total_composite/mean": 0.7625830173492432, "rewards/total_composite/std": 0.1522209644317627, "reward": 0.7625830173492432, "reward_std": 0.1522209495306015, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15767443180084229, "sampling/sampling_logp_difference/max": 1.4129719734191895, "sampling/importance_sampling_ratio/min": 0.24341876804828644, "sampling/importance_sampling_ratio/mean": 1.0165010690689087, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9005683064460754, "clip_ratio/low_mean": 0.030101103708148003, "clip_ratio/low_min": 0.030101103708148003, "clip_ratio/high_mean": 0.08492945320904255, "clip_ratio/high_max": 0.08492945320904255, "clip_ratio/region_mean": 0.11503055691719055, "reward_total_mean": 0.7625830173492432, "reward_meter_mean": 0.8724067211151123, "reward_meter_std": 0.3415348529815674, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4124999940395355, "reward_judge_quality_std": 0.1060660108923912, "reward_total_composite_mean": 0.7625830173492432, "reward_total_composite_std": 0.1522209644317627} {"timestamp_utc": "2026-04-13T02:30:11Z", "mode": "train", "global_step": 1670, "epoch": 0.167754897036665, "loss": 0.0317, "grad_norm": 14.129412651062012, "learning_rate": 4.942424242424243e-06, "num_tokens": 3014210.0, "completions/mean_length": 39.25, "completions/min_length": 37.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.25, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.7183547616004944, "rewards/meter/std": 0.35446882247924805, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9753074645996094, "rewards/repeat_soft/std": 0.019306225702166557, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.6979153752326965, "rewards/total_composite/std": 0.15711356699466705, "reward": 0.6979153752326965, "reward_std": 0.15711358189582825, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12266188114881516, "sampling/sampling_logp_difference/max": 1.4286894798278809, "sampling/importance_sampling_ratio/min": 0.23962275683879852, "sampling/importance_sampling_ratio/mean": 1.0225321054458618, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.68548384308815, "clip_ratio/low_mean": 0.041666668839752674, "clip_ratio/low_min": 0.041666668839752674, "clip_ratio/high_mean": 0.07591049186885357, "clip_ratio/high_max": 0.07591049186885357, "clip_ratio/region_mean": 0.11757716070860624, "reward_total_mean": 0.6979153752326965, "reward_meter_mean": 0.7183547616004944, "reward_meter_std": 0.35446882247924805, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9753074645996094, "reward_repeat_soft_std": 0.019306225702166557, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.6979153752326965, "reward_total_composite_std": 0.15711356699466705} {"timestamp_utc": "2026-04-13T02:30:20Z", "mode": "train", "global_step": 1671, "epoch": 0.16785534907081867, "loss": 0.0144, "grad_norm": 6.993900775909424, "learning_rate": 4.93939393939394e-06, "num_tokens": 3017008.0, "completions/mean_length": 166.75, "completions/min_length": 159.0, "completions/max_length": 181.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 166.75, "completions/min_terminated_length": 159.0, "completions/max_terminated_length": 181.0, "rewards/meter/mean": 0.5800771117210388, "rewards/meter/std": 0.3204021453857422, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9370157122612, "rewards/repeat_soft/std": 0.04991002008318901, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.12631452083587646, "rewards/total_composite/mean": 0.6041113138198853, "rewards/total_composite/std": 0.15444667637348175, "reward": 0.6041113138198853, "reward_std": 0.15444669127464294, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1415819525718689, "sampling/sampling_logp_difference/max": 3.002821683883667, "sampling/importance_sampling_ratio/min": 0.049646783620119095, "sampling/importance_sampling_ratio/mean": 1.0027042627334595, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.895163506269455, "clip_ratio/low_mean": 0.05738609191030264, "clip_ratio/low_min": 0.05738609191030264, "clip_ratio/high_mean": 0.0853233803063631, "clip_ratio/high_max": 0.0853233803063631, "clip_ratio/region_mean": 0.14270947221666574, "reward_total_mean": 0.6041113138198853, "reward_meter_mean": 0.5800771117210388, "reward_meter_std": 0.3204021453857422, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9370157122612, "reward_repeat_soft_std": 0.04991002008318901, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.12631452083587646, "reward_total_composite_mean": 0.6041113138198853, "reward_total_composite_std": 0.15444667637348175} {"timestamp_utc": "2026-04-13T02:30:26Z", "mode": "train", "global_step": 1672, "epoch": 0.16795580110497238, "loss": -0.0025, "grad_norm": 15.769043922424316, "learning_rate": 4.936363636363637e-06, "num_tokens": 3018368.0, "completions/mean_length": 27.0, "completions/min_length": 24.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.0, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.8071669340133667, "rewards/meter/std": 0.25294479727745056, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9431272745132446, "rewards/repeat_soft/std": 0.020711323246359825, "rewards/judge_quality/mean": 0.6262500286102295, "rewards/judge_quality/std": 0.2432481348514557, "rewards/total_composite/mean": 0.7954128384590149, "rewards/total_composite/std": 0.134674072265625, "reward": 0.7954128384590149, "reward_std": 0.1346740424633026, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13545671105384827, "sampling/sampling_logp_difference/max": 1.516526222229004, "sampling/importance_sampling_ratio/min": 0.2194729745388031, "sampling/importance_sampling_ratio/mean": 0.9932555556297302, "sampling/importance_sampling_ratio/max": 1.8970054388046265, "entropy": 0.7056831978261471, "clip_ratio/low_mean": 0.027354111894965172, "clip_ratio/low_min": 0.027354111894965172, "clip_ratio/high_mean": 0.09691772982478142, "clip_ratio/high_max": 0.09691772982478142, "clip_ratio/region_mean": 0.12427184171974659, "reward_total_mean": 0.7954128384590149, "reward_meter_mean": 0.8071669340133667, "reward_meter_std": 0.25294479727745056, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9431272745132446, "reward_repeat_soft_std": 0.020711323246359825, "reward_judge_quality_mean": 0.6262500286102295, "reward_judge_quality_std": 0.2432481348514557, "reward_total_composite_mean": 0.7954128384590149, "reward_total_composite_std": 0.134674072265625} {"timestamp_utc": "2026-04-13T02:30:38Z", "mode": "train", "global_step": 1673, "epoch": 0.16805625313912606, "loss": -0.1498, "grad_norm": 3.60160756111145, "learning_rate": 4.933333333333334e-06, "num_tokens": 3020220.0, "completions/mean_length": 127.5, "completions/min_length": 66.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 72.5714340209961, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.7547805309295654, "rewards/meter/std": 0.31769317388534546, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8801878690719604, "rewards/repeat_soft/std": 0.06324945390224457, "rewards/judge_quality/mean": 0.45625001192092896, "rewards/judge_quality/std": 0.26624035835266113, "rewards/total_composite/mean": 0.6488112807273865, "rewards/total_composite/std": 0.3140396475791931, "reward": 0.6488112807273865, "reward_std": 0.3140396475791931, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09648318588733673, "sampling/sampling_logp_difference/max": 1.2637972831726074, "sampling/importance_sampling_ratio/min": 0.2825789451599121, "sampling/importance_sampling_ratio/mean": 1.0119049549102783, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5316581502556801, "clip_ratio/low_mean": 0.027761456556618214, "clip_ratio/low_min": 0.027761456556618214, "clip_ratio/high_mean": 0.06057733716443181, "clip_ratio/high_max": 0.06057733716443181, "clip_ratio/region_mean": 0.08833879372105002, "reward_total_mean": 0.6488112807273865, "reward_meter_mean": 0.7547805309295654, "reward_meter_std": 0.31769317388534546, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8801878690719604, "reward_repeat_soft_std": 0.06324945390224457, "reward_judge_quality_mean": 0.45625001192092896, "reward_judge_quality_std": 0.26624035835266113, "reward_total_composite_mean": 0.6488112807273865, "reward_total_composite_std": 0.3140396475791931} {"timestamp_utc": "2026-04-13T02:30:49Z", "mode": "train", "global_step": 1674, "epoch": 0.16815670517327977, "loss": -0.1404, "grad_norm": 3.8145089149475098, "learning_rate": 4.9303030303030305e-06, "num_tokens": 3021951.0, "completions/mean_length": 121.375, "completions/min_length": 53.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 65.5714340209961, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.5069210529327393, "rewards/meter/std": 0.3460422456264496, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9787720441818237, "rewards/repeat_soft/std": 0.020404625684022903, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.547228217124939, "rewards/total_composite/std": 0.25089454650878906, "reward": 0.547228217124939, "reward_std": 0.25089454650878906, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14307934045791626, "sampling/sampling_logp_difference/max": 1.4295318126678467, "sampling/importance_sampling_ratio/min": 0.23942101001739502, "sampling/importance_sampling_ratio/mean": 1.0071113109588623, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6116698160767555, "clip_ratio/low_mean": 0.018939394503831863, "clip_ratio/low_min": 0.018939394503831863, "clip_ratio/high_mean": 0.11680454760789871, "clip_ratio/high_max": 0.11680454760789871, "clip_ratio/region_mean": 0.13574394211173058, "reward_total_mean": 0.547228217124939, "reward_meter_mean": 0.5069210529327393, "reward_meter_std": 0.3460422456264496, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9787720441818237, "reward_repeat_soft_std": 0.020404625684022903, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.547228217124939, "reward_total_composite_std": 0.25089454650878906} {"timestamp_utc": "2026-04-13T02:31:01Z", "mode": "train", "global_step": 1675, "epoch": 0.16825715720743345, "loss": -0.2212, "grad_norm": 1.9857542514801025, "learning_rate": 4.927272727272728e-06, "num_tokens": 3024270.0, "completions/mean_length": 287.875, "completions/min_length": 136.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 153.40000915527344, "completions/min_terminated_length": 136.0, "completions/max_terminated_length": 170.0, "rewards/meter/mean": 0.7504442930221558, "rewards/meter/std": 0.3732742369174957, "rewards/count_adherence/mean": 0.824999988079071, "rewards/count_adherence/std": 0.345377653837204, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.94571852684021, "rewards/repeat_soft/std": 0.043508727103471756, "rewards/judge_quality/mean": 0.11249999701976776, "rewards/judge_quality/std": 0.051754921674728394, "rewards/total_composite/mean": 0.491771399974823, "rewards/total_composite/std": 0.3243246376514435, "reward": 0.491771399974823, "reward_std": 0.3243246078491211, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13211014866828918, "sampling/sampling_logp_difference/max": 1.5731220245361328, "sampling/importance_sampling_ratio/min": 0.20739667117595673, "sampling/importance_sampling_ratio/mean": 1.0174709558486938, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5978853777050972, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08115033339709044, "clip_ratio/high_max": 0.08115033339709044, "clip_ratio/region_mean": 0.08115033339709044, "reward_total_mean": 0.491771399974823, "reward_meter_mean": 0.7504442930221558, "reward_meter_std": 0.3732742369174957, "reward_count_adherence_mean": 0.824999988079071, "reward_count_adherence_std": 0.345377653837204, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.94571852684021, "reward_repeat_soft_std": 0.043508727103471756, "reward_judge_quality_mean": 0.11249999701976776, "reward_judge_quality_std": 0.051754921674728394, "reward_total_composite_mean": 0.491771399974823, "reward_total_composite_std": 0.3243246376514435} {"timestamp_utc": "2026-04-13T02:31:09Z", "mode": "train", "global_step": 1676, "epoch": 0.16835760924158713, "loss": 0.0736, "grad_norm": 6.674496173858643, "learning_rate": 4.924242424242425e-06, "num_tokens": 3027253.0, "completions/mean_length": 172.875, "completions/min_length": 142.0, "completions/max_length": 199.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 172.875, "completions/min_terminated_length": 142.0, "completions/max_terminated_length": 199.0, "rewards/meter/mean": 0.9259616732597351, "rewards/meter/std": 0.16413716971874237, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9083374738693237, "rewards/repeat_soft/std": 0.050223615020513535, "rewards/judge_quality/mean": 0.24250000715255737, "rewards/judge_quality/std": 0.11792854964733124, "rewards/total_composite/mean": 0.7227665185928345, "rewards/total_composite/std": 0.08843529969453812, "reward": 0.7227665185928345, "reward_std": 0.08843529224395752, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11597383767366409, "sampling/sampling_logp_difference/max": 1.8806419372558594, "sampling/importance_sampling_ratio/min": 0.15249218046665192, "sampling/importance_sampling_ratio/mean": 1.0029966831207275, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6949771791696548, "clip_ratio/low_mean": 0.03702512662857771, "clip_ratio/low_min": 0.03702512662857771, "clip_ratio/high_mean": 0.08009605668485165, "clip_ratio/high_max": 0.08009605668485165, "clip_ratio/region_mean": 0.11712118331342936, "reward_total_mean": 0.7227665185928345, "reward_meter_mean": 0.9259616732597351, "reward_meter_std": 0.16413716971874237, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9083374738693237, "reward_repeat_soft_std": 0.050223615020513535, "reward_judge_quality_mean": 0.24250000715255737, "reward_judge_quality_std": 0.11792854964733124, "reward_total_composite_mean": 0.7227665185928345, "reward_total_composite_std": 0.08843529969453812} {"timestamp_utc": "2026-04-13T02:31:16Z", "mode": "train", "global_step": 1677, "epoch": 0.16845806127574084, "loss": 0.1437, "grad_norm": 11.21014404296875, "learning_rate": 4.9212121212121214e-06, "num_tokens": 3028992.0, "completions/mean_length": 56.375, "completions/min_length": 43.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.375, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9847921133041382, "rewards/meter/std": 0.005151356570422649, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9514276385307312, "rewards/repeat_soft/std": 0.07345323264598846, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.7914242148399353, "rewards/total_composite/std": 0.03895198926329613, "reward": 0.7914242148399353, "reward_std": 0.038951992988586426, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13904975354671478, "sampling/sampling_logp_difference/max": 1.4854178428649902, "sampling/importance_sampling_ratio/min": 0.2264077216386795, "sampling/importance_sampling_ratio/mean": 0.997774064540863, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8545645847916603, "clip_ratio/low_mean": 0.030497439671307802, "clip_ratio/low_min": 0.030497439671307802, "clip_ratio/high_mean": 0.08827344421297312, "clip_ratio/high_max": 0.08827344421297312, "clip_ratio/region_mean": 0.11877088388428092, "reward_total_mean": 0.7914242148399353, "reward_meter_mean": 0.9847921133041382, "reward_meter_std": 0.005151356570422649, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9514276385307312, "reward_repeat_soft_std": 0.07345323264598846, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.7914242148399353, "reward_total_composite_std": 0.03895198926329613} {"timestamp_utc": "2026-04-13T02:31:29Z", "mode": "train", "global_step": 1678, "epoch": 0.16855851330989452, "loss": -0.246, "grad_norm": 1.724330186843872, "learning_rate": 4.918181818181819e-06, "num_tokens": 3031873.0, "completions/mean_length": 243.125, "completions/min_length": 190.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 204.71429443359375, "completions/min_terminated_length": 190.0, "completions/max_terminated_length": 228.0, "rewards/meter/mean": 0.7497921586036682, "rewards/meter/std": 0.3468679189682007, "rewards/count_adherence/mean": 0.8541666269302368, "rewards/count_adherence/std": 0.3500283658504486, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8363153338432312, "rewards/repeat_soft/std": 0.11198506504297256, "rewards/judge_quality/mean": 0.20874999463558197, "rewards/judge_quality/std": 0.11038083583116531, "rewards/total_composite/mean": 0.5974129438400269, "rewards/total_composite/std": 0.24952706694602966, "reward": 0.5974129438400269, "reward_std": 0.24952705204486847, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10876287519931793, "sampling/sampling_logp_difference/max": 2.101405143737793, "sampling/importance_sampling_ratio/min": 0.12228447943925858, "sampling/importance_sampling_ratio/mean": 1.0120439529418945, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.699626088142395, "clip_ratio/low_mean": 0.009703196585178375, "clip_ratio/low_min": 0.009703196585178375, "clip_ratio/high_mean": 0.07270438969135284, "clip_ratio/high_max": 0.07270438969135284, "clip_ratio/region_mean": 0.08240758627653122, "reward_total_mean": 0.5974129438400269, "reward_meter_mean": 0.7497921586036682, "reward_meter_std": 0.3468679189682007, "reward_count_adherence_mean": 0.8541666269302368, "reward_count_adherence_std": 0.3500283658504486, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8363153338432312, "reward_repeat_soft_std": 0.11198506504297256, "reward_judge_quality_mean": 0.20874999463558197, "reward_judge_quality_std": 0.11038083583116531, "reward_total_composite_mean": 0.5974129438400269, "reward_total_composite_std": 0.24952706694602966} {"timestamp_utc": "2026-04-13T02:31:37Z", "mode": "train", "global_step": 1679, "epoch": 0.16865896534404823, "loss": 0.1097, "grad_norm": 5.888236999511719, "learning_rate": 4.915151515151516e-06, "num_tokens": 3034612.0, "completions/mean_length": 163.375, "completions/min_length": 137.0, "completions/max_length": 200.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 163.375, "completions/min_terminated_length": 137.0, "completions/max_terminated_length": 200.0, "rewards/meter/mean": 0.9077426195144653, "rewards/meter/std": 0.20689477026462555, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.908366322517395, "rewards/repeat_soft/std": 0.04951225966215134, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.12772038578987122, "rewards/total_composite/mean": 0.7561958432197571, "rewards/total_composite/std": 0.12386998534202576, "reward": 0.7561958432197571, "reward_std": 0.12387000024318695, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11018289625644684, "sampling/sampling_logp_difference/max": 1.679776668548584, "sampling/importance_sampling_ratio/min": 0.18641559779644012, "sampling/importance_sampling_ratio/mean": 1.0051729679107666, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6841397061944008, "clip_ratio/low_mean": 0.01881983270868659, "clip_ratio/low_min": 0.01881983270868659, "clip_ratio/high_mean": 0.09441139362752438, "clip_ratio/high_max": 0.09441139362752438, "clip_ratio/region_mean": 0.11323122633621097, "reward_total_mean": 0.7561958432197571, "reward_meter_mean": 0.9077426195144653, "reward_meter_std": 0.20689477026462555, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.908366322517395, "reward_repeat_soft_std": 0.04951225966215134, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.12772038578987122, "reward_total_composite_mean": 0.7561958432197571, "reward_total_composite_std": 0.12386998534202576} {"timestamp_utc": "2026-04-13T02:31:45Z", "mode": "train", "global_step": 1680, "epoch": 0.1687594173782019, "loss": 0.0475, "grad_norm": 11.465644836425781, "learning_rate": 4.912121212121212e-06, "num_tokens": 3036627.0, "completions/mean_length": 89.875, "completions/min_length": 76.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.875, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.4340985417366028, "rewards/meter/std": 0.37435710430145264, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.974621057510376, "rewards/repeat_soft/std": 0.02853590063750744, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.5616813898086548, "rewards/total_composite/std": 0.16492106020450592, "reward": 0.5616813898086548, "reward_std": 0.16492104530334473, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1382889598608017, "sampling/sampling_logp_difference/max": 1.9854683876037598, "sampling/importance_sampling_ratio/min": 0.1373162865638733, "sampling/importance_sampling_ratio/mean": 0.9987658858299255, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6624528467655182, "clip_ratio/low_mean": 0.07675399631261826, "clip_ratio/low_min": 0.07675399631261826, "clip_ratio/high_mean": 0.04738641995936632, "clip_ratio/high_max": 0.04738641995936632, "clip_ratio/region_mean": 0.12414041627198458, "reward_total_mean": 0.5616813898086548, "reward_meter_mean": 0.4340985417366028, "reward_meter_std": 0.37435710430145264, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.974621057510376, "reward_repeat_soft_std": 0.02853590063750744, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.5616813898086548, "reward_total_composite_std": 0.16492106020450592} {"timestamp_utc": "2026-04-13T02:31:51Z", "mode": "train", "global_step": 1681, "epoch": 0.1688598694123556, "loss": 0.0226, "grad_norm": 11.642288208007812, "learning_rate": 4.90909090909091e-06, "num_tokens": 3038361.0, "completions/mean_length": 61.75, "completions/min_length": 55.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.75, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.8693984746932983, "rewards/meter/std": 0.29402217268943787, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9645372629165649, "rewards/repeat_soft/std": 0.03503952920436859, "rewards/judge_quality/mean": 0.42124998569488525, "rewards/judge_quality/std": 0.06998724490404129, "rewards/total_composite/mean": 0.7640580534934998, "rewards/total_composite/std": 0.12981751561164856, "reward": 0.7640580534934998, "reward_std": 0.12981751561164856, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13576559722423553, "sampling/sampling_logp_difference/max": 1.6464147567749023, "sampling/importance_sampling_ratio/min": 0.19273969531059265, "sampling/importance_sampling_ratio/mean": 1.0154590606689453, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8473916202783585, "clip_ratio/low_mean": 0.03390151634812355, "clip_ratio/low_min": 0.03390151634812355, "clip_ratio/high_mean": 0.1072694743052125, "clip_ratio/high_max": 0.1072694743052125, "clip_ratio/region_mean": 0.14117099065333605, "reward_total_mean": 0.7640580534934998, "reward_meter_mean": 0.8693984746932983, "reward_meter_std": 0.29402217268943787, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9645372629165649, "reward_repeat_soft_std": 0.03503952920436859, "reward_judge_quality_mean": 0.42124998569488525, "reward_judge_quality_std": 0.06998724490404129, "reward_total_composite_mean": 0.7640580534934998, "reward_total_composite_std": 0.12981751561164856} {"timestamp_utc": "2026-04-13T02:31:58Z", "mode": "train", "global_step": 1682, "epoch": 0.1689603214465093, "loss": -0.0056, "grad_norm": 15.687945365905762, "learning_rate": 4.906060606060606e-06, "num_tokens": 3039960.0, "completions/mean_length": 48.875, "completions/min_length": 43.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.875, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9845795631408691, "rewards/meter/std": 0.009886974468827248, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9075263738632202, "rewards/repeat_soft/std": 0.0537593737244606, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8109384179115295, "rewards/total_composite/std": 0.008434985764324665, "reward": 0.8109384179115295, "reward_std": 0.008434985764324665, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10835865885019302, "sampling/sampling_logp_difference/max": 2.1667299270629883, "sampling/importance_sampling_ratio/min": 0.1145515963435173, "sampling/importance_sampling_ratio/mean": 1.005966305732727, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5870461314916611, "clip_ratio/low_mean": 0.04740891559049487, "clip_ratio/low_min": 0.04740891559049487, "clip_ratio/high_mean": 0.04437817446887493, "clip_ratio/high_max": 0.04437817446887493, "clip_ratio/region_mean": 0.0917870900593698, "reward_total_mean": 0.8109384179115295, "reward_meter_mean": 0.9845795631408691, "reward_meter_std": 0.009886974468827248, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9075263738632202, "reward_repeat_soft_std": 0.0537593737244606, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8109384179115295, "reward_total_composite_std": 0.008434985764324665} {"timestamp_utc": "2026-04-13T02:32:10Z", "mode": "train", "global_step": 1683, "epoch": 0.16906077348066298, "loss": -0.0938, "grad_norm": 1.930603265762329, "learning_rate": 4.903030303030303e-06, "num_tokens": 3041485.0, "completions/mean_length": 160.625, "completions/min_length": 37.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 43.5, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.24386516213417053, "rewards/meter/std": 0.25157612562179565, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9890428781509399, "rewards/repeat_soft/std": 0.01381877064704895, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.29609060287475586, "rewards/total_composite/mean": 0.4213522672653198, "rewards/total_composite/std": 0.28931474685668945, "reward": 0.4213522672653198, "reward_std": 0.28931471705436707, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14988073706626892, "sampling/sampling_logp_difference/max": 1.448014259338379, "sampling/importance_sampling_ratio/min": 0.2657962441444397, "sampling/importance_sampling_ratio/mean": 1.0223358869552612, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6398867592215538, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.11730421893298626, "clip_ratio/high_max": 0.11730421893298626, "clip_ratio/region_mean": 0.11730421893298626, "reward_total_mean": 0.4213522672653198, "reward_meter_mean": 0.24386516213417053, "reward_meter_std": 0.25157612562179565, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9890428781509399, "reward_repeat_soft_std": 0.01381877064704895, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.29609060287475586, "reward_total_composite_mean": 0.4213522672653198, "reward_total_composite_std": 0.28931474685668945} {"timestamp_utc": "2026-04-13T02:32:18Z", "mode": "train", "global_step": 1684, "epoch": 0.1691612255148167, "loss": -0.0434, "grad_norm": 19.372116088867188, "learning_rate": 4.9000000000000005e-06, "num_tokens": 3042841.0, "completions/mean_length": 30.5, "completions/min_length": 27.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.5, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.5630418062210083, "rewards/meter/std": 0.3793165683746338, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.956274688243866, "rewards/repeat_soft/std": 0.012105435132980347, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720350325107574, "rewards/total_composite/mean": 0.6647462844848633, "rewards/total_composite/std": 0.12174684554338455, "reward": 0.6647462844848633, "reward_std": 0.12174684554338455, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14271321892738342, "sampling/sampling_logp_difference/max": 1.2162787914276123, "sampling/importance_sampling_ratio/min": 0.2963308095932007, "sampling/importance_sampling_ratio/mean": 1.0222527980804443, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9309145510196686, "clip_ratio/low_mean": 0.05908978218212724, "clip_ratio/low_min": 0.05908978218212724, "clip_ratio/high_mean": 0.08585318177938461, "clip_ratio/high_max": 0.08585318177938461, "clip_ratio/region_mean": 0.14494296396151185, "reward_total_mean": 0.6647462844848633, "reward_meter_mean": 0.5630418062210083, "reward_meter_std": 0.3793165683746338, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.956274688243866, "reward_repeat_soft_std": 0.012105435132980347, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720350325107574, "reward_total_composite_mean": 0.6647462844848633, "reward_total_composite_std": 0.12174684554338455} {"timestamp_utc": "2026-04-13T02:32:27Z", "mode": "train", "global_step": 1685, "epoch": 0.16926167754897037, "loss": 0.1187, "grad_norm": 9.072885513305664, "learning_rate": 4.896969696969697e-06, "num_tokens": 3045279.0, "completions/mean_length": 128.75, "completions/min_length": 107.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.75, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.8229939341545105, "rewards/meter/std": 0.29144978523254395, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8926151990890503, "rewards/repeat_soft/std": 0.07545717805624008, "rewards/judge_quality/mean": 0.29750001430511475, "rewards/judge_quality/std": 0.1349867582321167, "rewards/total_composite/mean": 0.6988587975502014, "rewards/total_composite/std": 0.1477181762456894, "reward": 0.6988587975502014, "reward_std": 0.1477181762456894, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10665714740753174, "sampling/sampling_logp_difference/max": 1.5129337310791016, "sampling/importance_sampling_ratio/min": 0.22026284039020538, "sampling/importance_sampling_ratio/mean": 1.0134148597717285, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5732622481882572, "clip_ratio/low_mean": 0.01808035746216774, "clip_ratio/low_min": 0.01808035746216774, "clip_ratio/high_mean": 0.0769457584246993, "clip_ratio/high_max": 0.0769457584246993, "clip_ratio/region_mean": 0.09502611588686705, "reward_total_mean": 0.6988587975502014, "reward_meter_mean": 0.8229939341545105, "reward_meter_std": 0.29144978523254395, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8926151990890503, "reward_repeat_soft_std": 0.07545717805624008, "reward_judge_quality_mean": 0.29750001430511475, "reward_judge_quality_std": 0.1349867582321167, "reward_total_composite_mean": 0.6988587975502014, "reward_total_composite_std": 0.1477181762456894} {"timestamp_utc": "2026-04-13T02:32:36Z", "mode": "train", "global_step": 1686, "epoch": 0.16936212958312405, "loss": 0.0402, "grad_norm": 7.634072780609131, "learning_rate": 4.893939393939394e-06, "num_tokens": 3047920.0, "completions/mean_length": 134.125, "completions/min_length": 97.0, "completions/max_length": 154.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 134.125, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 154.0, "rewards/meter/mean": 0.5984695553779602, "rewards/meter/std": 0.33936166763305664, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9390931725502014, "rewards/repeat_soft/std": 0.022914685308933258, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.6163455843925476, "rewards/total_composite/std": 0.16010397672653198, "reward": 0.6163455843925476, "reward_std": 0.1601039469242096, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12741225957870483, "sampling/sampling_logp_difference/max": 2.5178401470184326, "sampling/importance_sampling_ratio/min": 0.08063358068466187, "sampling/importance_sampling_ratio/mean": 1.00706148147583, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6222167909145355, "clip_ratio/low_mean": 0.05994720011949539, "clip_ratio/low_min": 0.05994720011949539, "clip_ratio/high_mean": 0.06416744086891413, "clip_ratio/high_max": 0.06416744086891413, "clip_ratio/region_mean": 0.12411464098840952, "reward_total_mean": 0.6163455843925476, "reward_meter_mean": 0.5984695553779602, "reward_meter_std": 0.33936166763305664, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9390931725502014, "reward_repeat_soft_std": 0.022914685308933258, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.6163455843925476, "reward_total_composite_std": 0.16010397672653198} {"timestamp_utc": "2026-04-13T02:32:42Z", "mode": "train", "global_step": 1687, "epoch": 0.16946258161727776, "loss": 0.0384, "grad_norm": 9.91786003112793, "learning_rate": 4.8909090909090914e-06, "num_tokens": 3049746.0, "completions/mean_length": 57.25, "completions/min_length": 54.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.25, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.7774099111557007, "rewards/meter/std": 0.3028985261917114, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.937341034412384, "rewards/repeat_soft/std": 0.04964404180645943, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7206935882568359, "rewards/total_composite/std": 0.1375037282705307, "reward": 0.7206935882568359, "reward_std": 0.1375037282705307, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12247586995363235, "sampling/sampling_logp_difference/max": 1.7366056442260742, "sampling/importance_sampling_ratio/min": 0.1761171966791153, "sampling/importance_sampling_ratio/mean": 1.0022884607315063, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6591398790478706, "clip_ratio/low_mean": 0.03272026591002941, "clip_ratio/low_min": 0.03272026591002941, "clip_ratio/high_mean": 0.0815791105851531, "clip_ratio/high_max": 0.0815791105851531, "clip_ratio/region_mean": 0.11429937649518251, "reward_total_mean": 0.7206935882568359, "reward_meter_mean": 0.7774099111557007, "reward_meter_std": 0.3028985261917114, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.937341034412384, "reward_repeat_soft_std": 0.04964404180645943, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7206935882568359, "reward_total_composite_std": 0.1375037282705307} {"timestamp_utc": "2026-04-13T02:32:50Z", "mode": "train", "global_step": 1688, "epoch": 0.16956303365143144, "loss": -0.0038, "grad_norm": 4.496267795562744, "learning_rate": 4.887878787878788e-06, "num_tokens": 3052850.0, "completions/mean_length": 190.0, "completions/min_length": 172.0, "completions/max_length": 200.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 190.0, "completions/min_terminated_length": 172.0, "completions/max_terminated_length": 200.0, "rewards/meter/mean": 0.9834912419319153, "rewards/meter/std": 0.029435941949486732, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.719551682472229, "rewards/repeat_soft/std": 0.10000699013471603, "rewards/judge_quality/mean": 0.2887499928474426, "rewards/judge_quality/std": 0.11630470305681229, "rewards/total_composite/mean": 0.7480262517929077, "rewards/total_composite/std": 0.04160228744149208, "reward": 0.7480262517929077, "reward_std": 0.041602279990911484, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07856861501932144, "sampling/sampling_logp_difference/max": 2.341573715209961, "sampling/importance_sampling_ratio/min": 0.09617616981267929, "sampling/importance_sampling_ratio/mean": 1.0041481256484985, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.38900480046868324, "clip_ratio/low_mean": 0.030887856148183346, "clip_ratio/low_min": 0.030887856148183346, "clip_ratio/high_mean": 0.03155966568738222, "clip_ratio/high_max": 0.03155966568738222, "clip_ratio/region_mean": 0.06244752183556557, "reward_total_mean": 0.7480262517929077, "reward_meter_mean": 0.9834912419319153, "reward_meter_std": 0.029435941949486732, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.719551682472229, "reward_repeat_soft_std": 0.10000699013471603, "reward_judge_quality_mean": 0.2887499928474426, "reward_judge_quality_std": 0.11630470305681229, "reward_total_composite_mean": 0.7480262517929077, "reward_total_composite_std": 0.04160228744149208} {"timestamp_utc": "2026-04-13T02:32:57Z", "mode": "train", "global_step": 1689, "epoch": 0.16966348568558512, "loss": 0.0338, "grad_norm": 10.645234107971191, "learning_rate": 4.884848484848485e-06, "num_tokens": 3054629.0, "completions/mean_length": 63.375, "completions/min_length": 56.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.375, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9894886016845703, "rewards/meter/std": 0.009210045449435711, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9813227653503418, "rewards/repeat_soft/std": 0.01925262250006199, "rewards/judge_quality/mean": 0.6100000143051147, "rewards/judge_quality/std": 0.24628673493862152, "rewards/total_composite/mean": 0.8764021396636963, "rewards/total_composite/std": 0.07458680868148804, "reward": 0.8764021396636963, "reward_std": 0.07458680123090744, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11190931499004364, "sampling/sampling_logp_difference/max": 0.978126049041748, "sampling/importance_sampling_ratio/min": 0.3760150969028473, "sampling/importance_sampling_ratio/mean": 0.996849775314331, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6611417010426521, "clip_ratio/low_mean": 0.06120815686881542, "clip_ratio/low_min": 0.06120815686881542, "clip_ratio/high_mean": 0.05589453782886267, "clip_ratio/high_max": 0.05589453782886267, "clip_ratio/region_mean": 0.11710269469767809, "reward_total_mean": 0.8764021396636963, "reward_meter_mean": 0.9894886016845703, "reward_meter_std": 0.009210045449435711, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9813227653503418, "reward_repeat_soft_std": 0.01925262250006199, "reward_judge_quality_mean": 0.6100000143051147, "reward_judge_quality_std": 0.24628673493862152, "reward_total_composite_mean": 0.8764021396636963, "reward_total_composite_std": 0.07458680868148804} {"timestamp_utc": "2026-04-13T02:33:09Z", "mode": "train", "global_step": 1690, "epoch": 0.16976393771973883, "loss": -0.0395, "grad_norm": 4.22714900970459, "learning_rate": 4.881818181818182e-06, "num_tokens": 3056158.0, "completions/mean_length": 97.125, "completions/min_length": 26.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 37.85714340209961, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.7405763864517212, "rewards/meter/std": 0.45406463742256165, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9664591550827026, "rewards/repeat_soft/std": 0.023397037759423256, "rewards/judge_quality/mean": 0.33249998092651367, "rewards/judge_quality/std": 0.1580009013414383, "rewards/total_composite/mean": 0.6484053134918213, "rewards/total_composite/std": 0.3117710053920746, "reward": 0.6484053134918213, "reward_std": 0.3117710053920746, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17454852163791656, "sampling/sampling_logp_difference/max": 1.8622236251831055, "sampling/importance_sampling_ratio/min": 0.15532685816287994, "sampling/importance_sampling_ratio/mean": 1.0369752645492554, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9463625475764275, "clip_ratio/low_mean": 0.02459016442298889, "clip_ratio/low_min": 0.02459016442298889, "clip_ratio/high_mean": 0.13620835542678833, "clip_ratio/high_max": 0.13620835542678833, "clip_ratio/region_mean": 0.16079851984977722, "reward_total_mean": 0.6484053134918213, "reward_meter_mean": 0.7405763864517212, "reward_meter_std": 0.45406463742256165, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9664591550827026, "reward_repeat_soft_std": 0.023397037759423256, "reward_judge_quality_mean": 0.33249998092651367, "reward_judge_quality_std": 0.1580009013414383, "reward_total_composite_mean": 0.6484053134918213, "reward_total_composite_std": 0.3117710053920746} {"timestamp_utc": "2026-04-13T02:33:15Z", "mode": "train", "global_step": 1691, "epoch": 0.1698643897538925, "loss": 0.0292, "grad_norm": 13.561014175415039, "learning_rate": 4.878787878787879e-06, "num_tokens": 3057872.0, "completions/mean_length": 56.25, "completions/min_length": 45.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.25, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.621579647064209, "rewards/meter/std": 0.39828261733055115, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9705144166946411, "rewards/repeat_soft/std": 0.020721375942230225, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.6527622938156128, "rewards/total_composite/std": 0.17915360629558563, "reward": 0.6527622938156128, "reward_std": 0.17915359139442444, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17151528596878052, "sampling/sampling_logp_difference/max": 2.507054090499878, "sampling/importance_sampling_ratio/min": 0.08150800317525864, "sampling/importance_sampling_ratio/mean": 0.9808043241500854, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5159989967942238, "clip_ratio/low_mean": 0.06147496495395899, "clip_ratio/low_min": 0.06147496495395899, "clip_ratio/high_mean": 0.10093087144196033, "clip_ratio/high_max": 0.10093087144196033, "clip_ratio/region_mean": 0.16240583639591932, "reward_total_mean": 0.6527622938156128, "reward_meter_mean": 0.621579647064209, "reward_meter_std": 0.39828261733055115, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9705144166946411, "reward_repeat_soft_std": 0.020721375942230225, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.6527622938156128, "reward_total_composite_std": 0.17915360629558563} {"timestamp_utc": "2026-04-13T02:33:23Z", "mode": "train", "global_step": 1692, "epoch": 0.16996484178804622, "loss": 0.0045, "grad_norm": 6.864474296569824, "learning_rate": 4.875757575757576e-06, "num_tokens": 3060004.0, "completions/mean_length": 97.5, "completions/min_length": 94.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.5, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.9759260416030884, "rewards/meter/std": 0.017225569114089012, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9192461967468262, "rewards/repeat_soft/std": 0.04964800179004669, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.7969663143157959, "rewards/total_composite/std": 0.030293544754385948, "reward": 0.7969663143157959, "reward_std": 0.030293535441160202, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12392042577266693, "sampling/sampling_logp_difference/max": 1.6576030254364014, "sampling/importance_sampling_ratio/min": 0.19059528410434723, "sampling/importance_sampling_ratio/mean": 1.0266973972320557, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7421442568302155, "clip_ratio/low_mean": 0.025963102467358112, "clip_ratio/low_min": 0.025963102467358112, "clip_ratio/high_mean": 0.08809616230428219, "clip_ratio/high_max": 0.08809616230428219, "clip_ratio/region_mean": 0.1140592647716403, "reward_total_mean": 0.7969663143157959, "reward_meter_mean": 0.9759260416030884, "reward_meter_std": 0.017225569114089012, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9192461967468262, "reward_repeat_soft_std": 0.04964800179004669, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.7969663143157959, "reward_total_composite_std": 0.030293544754385948} {"timestamp_utc": "2026-04-13T02:33:30Z", "mode": "train", "global_step": 1693, "epoch": 0.1700652938221999, "loss": 0.0109, "grad_norm": 6.08503532409668, "learning_rate": 4.872727272727273e-06, "num_tokens": 3062871.0, "completions/mean_length": 156.375, "completions/min_length": 146.0, "completions/max_length": 169.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 156.375, "completions/min_terminated_length": 146.0, "completions/max_terminated_length": 169.0, "rewards/meter/mean": 0.7394919395446777, "rewards/meter/std": 0.3314027190208435, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8822628855705261, "rewards/repeat_soft/std": 0.05251713842153549, "rewards/judge_quality/mean": 0.20875000953674316, "rewards/judge_quality/std": 0.09657527506351471, "rewards/total_composite/mean": 0.6336226463317871, "rewards/total_composite/std": 0.1641036570072174, "reward": 0.6336226463317871, "reward_std": 0.16410362720489502, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10209805518388748, "sampling/sampling_logp_difference/max": 2.9461166858673096, "sampling/importance_sampling_ratio/min": 0.05254335328936577, "sampling/importance_sampling_ratio/mean": 1.0112775564193726, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6130328848958015, "clip_ratio/low_mean": 0.028994388412684202, "clip_ratio/low_min": 0.028994388412684202, "clip_ratio/high_mean": 0.06361437402665615, "clip_ratio/high_max": 0.06361437402665615, "clip_ratio/region_mean": 0.09260876243934035, "reward_total_mean": 0.6336226463317871, "reward_meter_mean": 0.7394919395446777, "reward_meter_std": 0.3314027190208435, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8822628855705261, "reward_repeat_soft_std": 0.05251713842153549, "reward_judge_quality_mean": 0.20875000953674316, "reward_judge_quality_std": 0.09657527506351471, "reward_total_composite_mean": 0.6336226463317871, "reward_total_composite_std": 0.1641036570072174} {"timestamp_utc": "2026-04-13T02:33:36Z", "mode": "train", "global_step": 1694, "epoch": 0.17016574585635358, "loss": 0.0475, "grad_norm": 13.362269401550293, "learning_rate": 4.8696969696969705e-06, "num_tokens": 3064646.0, "completions/mean_length": 38.875, "completions/min_length": 37.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.875, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.7085868120193481, "rewards/meter/std": 0.3481637239456177, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9549683928489685, "rewards/repeat_soft/std": 0.04819221794605255, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.7113609313964844, "rewards/total_composite/std": 0.17452330887317657, "reward": 0.7113609313964844, "reward_std": 0.17452330887317657, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1312226504087448, "sampling/sampling_logp_difference/max": 1.3601198196411133, "sampling/importance_sampling_ratio/min": 0.2566300332546234, "sampling/importance_sampling_ratio/mean": 0.9861375689506531, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6458666399121284, "clip_ratio/low_mean": 0.03324622567743063, "clip_ratio/low_min": 0.03324622567743063, "clip_ratio/high_mean": 0.10530435852706432, "clip_ratio/high_max": 0.10530435852706432, "clip_ratio/region_mean": 0.13855058420449495, "reward_total_mean": 0.7113609313964844, "reward_meter_mean": 0.7085868120193481, "reward_meter_std": 0.3481637239456177, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9549683928489685, "reward_repeat_soft_std": 0.04819221794605255, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.7113609313964844, "reward_total_composite_std": 0.17452330887317657} {"timestamp_utc": "2026-04-13T02:33:43Z", "mode": "train", "global_step": 1695, "epoch": 0.1702661978905073, "loss": 0.0248, "grad_norm": 13.374736785888672, "learning_rate": 4.866666666666667e-06, "num_tokens": 3066239.0, "completions/mean_length": 40.125, "completions/min_length": 36.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.125, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9284267425537109, "rewards/meter/std": 0.05355209484696388, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9847977161407471, "rewards/repeat_soft/std": 0.018957821652293205, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.8046467900276184, "rewards/total_composite/std": 0.06929448992013931, "reward": 0.8046467900276184, "reward_std": 0.06929447501897812, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10006143152713776, "sampling/sampling_logp_difference/max": 1.2706167697906494, "sampling/importance_sampling_ratio/min": 0.28065845370292664, "sampling/importance_sampling_ratio/mean": 1.0131535530090332, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6489324793219566, "clip_ratio/low_mean": 0.04613095335662365, "clip_ratio/low_min": 0.04613095335662365, "clip_ratio/high_mean": 0.06490803509950638, "clip_ratio/high_max": 0.06490803509950638, "clip_ratio/region_mean": 0.11103898845613003, "reward_total_mean": 0.8046467900276184, "reward_meter_mean": 0.9284267425537109, "reward_meter_std": 0.05355209484696388, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9847977161407471, "reward_repeat_soft_std": 0.018957821652293205, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.8046467900276184, "reward_total_composite_std": 0.06929448992013931} {"timestamp_utc": "2026-04-13T02:33:52Z", "mode": "train", "global_step": 1696, "epoch": 0.17036664992466097, "loss": -0.0213, "grad_norm": 8.801530838012695, "learning_rate": 4.863636363636364e-06, "num_tokens": 3068520.0, "completions/mean_length": 102.125, "completions/min_length": 83.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.125, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.9153744578361511, "rewards/meter/std": 0.08265617489814758, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8363927602767944, "rewards/repeat_soft/std": 0.07219633460044861, "rewards/judge_quality/mean": 0.2212499976158142, "rewards/judge_quality/std": 0.09433034062385559, "rewards/total_composite/mean": 0.7119327783584595, "rewards/total_composite/std": 0.049712274223566055, "reward": 0.7119327783584595, "reward_std": 0.04971226677298546, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09567604213953018, "sampling/sampling_logp_difference/max": 1.7773053646087646, "sampling/importance_sampling_ratio/min": 0.16909317672252655, "sampling/importance_sampling_ratio/mean": 1.0112688541412354, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5565677136182785, "clip_ratio/low_mean": 0.05930054374039173, "clip_ratio/low_min": 0.05930054374039173, "clip_ratio/high_mean": 0.03723760601133108, "clip_ratio/high_max": 0.03723760601133108, "clip_ratio/region_mean": 0.09653814975172281, "reward_total_mean": 0.7119327783584595, "reward_meter_mean": 0.9153744578361511, "reward_meter_std": 0.08265617489814758, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8363927602767944, "reward_repeat_soft_std": 0.07219633460044861, "reward_judge_quality_mean": 0.2212499976158142, "reward_judge_quality_std": 0.09433034062385559, "reward_total_composite_mean": 0.7119327783584595, "reward_total_composite_std": 0.049712274223566055} {"timestamp_utc": "2026-04-13T02:34:00Z", "mode": "train", "global_step": 1697, "epoch": 0.17046710195881468, "loss": 0.0405, "grad_norm": 5.5018157958984375, "learning_rate": 4.8606060606060615e-06, "num_tokens": 3070647.0, "completions/mean_length": 102.875, "completions/min_length": 96.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.875, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.9157636165618896, "rewards/meter/std": 0.08571980148553848, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7037443518638611, "rewards/repeat_soft/std": 0.13367852568626404, "rewards/judge_quality/mean": 0.16250000894069672, "rewards/judge_quality/std": 0.0353553369641304, "rewards/total_composite/mean": 0.6812180876731873, "rewards/total_composite/std": 0.05267798900604248, "reward": 0.6812180876731873, "reward_std": 0.052677981555461884, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08133301883935928, "sampling/sampling_logp_difference/max": 4.7354044914245605, "sampling/importance_sampling_ratio/min": 0.008778897114098072, "sampling/importance_sampling_ratio/mean": 1.004461407661438, "sampling/importance_sampling_ratio/max": 1.9819973707199097, "entropy": 0.3707086406648159, "clip_ratio/low_mean": 0.023616482270881534, "clip_ratio/low_min": 0.023616482270881534, "clip_ratio/high_mean": 0.04488610429689288, "clip_ratio/high_max": 0.04488610429689288, "clip_ratio/region_mean": 0.06850258656777442, "reward_total_mean": 0.6812180876731873, "reward_meter_mean": 0.9157636165618896, "reward_meter_std": 0.08571980148553848, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7037443518638611, "reward_repeat_soft_std": 0.13367852568626404, "reward_judge_quality_mean": 0.16250000894069672, "reward_judge_quality_std": 0.0353553369641304, "reward_total_composite_mean": 0.6812180876731873, "reward_total_composite_std": 0.05267798900604248} {"timestamp_utc": "2026-04-13T02:34:08Z", "mode": "train", "global_step": 1698, "epoch": 0.17056755399296836, "loss": 0.015, "grad_norm": 13.367608070373535, "learning_rate": 4.857575757575758e-06, "num_tokens": 3072486.0, "completions/mean_length": 60.875, "completions/min_length": 55.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.7485918402671814, "rewards/meter/std": 0.31665173172950745, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9272580146789551, "rewards/repeat_soft/std": 0.027281228452920914, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.7288421392440796, "rewards/total_composite/std": 0.16666406393051147, "reward": 0.7288421392440796, "reward_std": 0.16666406393051147, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13003471493721008, "sampling/sampling_logp_difference/max": 1.6208226680755615, "sampling/importance_sampling_ratio/min": 0.1977359503507614, "sampling/importance_sampling_ratio/mean": 1.002998948097229, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5984363332390785, "clip_ratio/low_mean": 0.04061114136129618, "clip_ratio/low_min": 0.04061114136129618, "clip_ratio/high_mean": 0.08967244299128652, "clip_ratio/high_max": 0.08967244299128652, "clip_ratio/region_mean": 0.1302835843525827, "reward_total_mean": 0.7288421392440796, "reward_meter_mean": 0.7485918402671814, "reward_meter_std": 0.31665173172950745, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9272580146789551, "reward_repeat_soft_std": 0.027281228452920914, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.7288421392440796, "reward_total_composite_std": 0.16666406393051147} {"timestamp_utc": "2026-04-13T02:34:16Z", "mode": "train", "global_step": 1699, "epoch": 0.17066800602712204, "loss": 0.0929, "grad_norm": 9.515035629272461, "learning_rate": 4.854545454545455e-06, "num_tokens": 3074199.0, "completions/mean_length": 58.125, "completions/min_length": 52.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.125, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.7407875061035156, "rewards/meter/std": 0.43322673439979553, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9135535359382629, "rewards/repeat_soft/std": 0.09290007501840591, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.0843462198972702, "rewards/total_composite/mean": 0.6902097463607788, "rewards/total_composite/std": 0.19789040088653564, "reward": 0.6902097463607788, "reward_std": 0.19789038598537445, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11674092710018158, "sampling/sampling_logp_difference/max": 1.7875111103057861, "sampling/importance_sampling_ratio/min": 0.16737623512744904, "sampling/importance_sampling_ratio/mean": 0.9886609315872192, "sampling/importance_sampling_ratio/max": 1.807699203491211, "entropy": 0.4801136925816536, "clip_ratio/low_mean": 0.020729638636112213, "clip_ratio/low_min": 0.020729638636112213, "clip_ratio/high_mean": 0.07282136753201485, "clip_ratio/high_max": 0.07282136753201485, "clip_ratio/region_mean": 0.09355100616812706, "reward_total_mean": 0.6902097463607788, "reward_meter_mean": 0.7407875061035156, "reward_meter_std": 0.43322673439979553, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9135535359382629, "reward_repeat_soft_std": 0.09290007501840591, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.0843462198972702, "reward_total_composite_mean": 0.6902097463607788, "reward_total_composite_std": 0.19789040088653564} {"timestamp_utc": "2026-04-13T02:34:25Z", "mode": "train", "global_step": 1700, "epoch": 0.17076845806127575, "loss": 0.0544, "grad_norm": 7.078593730926514, "learning_rate": 4.851515151515152e-06, "num_tokens": 3076495.0, "completions/mean_length": 120.0, "completions/min_length": 105.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.0, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.6779184341430664, "rewards/meter/std": 0.355674147605896, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9150282740592957, "rewards/repeat_soft/std": 0.05642801895737648, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.6534411907196045, "rewards/total_composite/std": 0.15114767849445343, "reward": 0.6534411907196045, "reward_std": 0.15114766359329224, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10782790929079056, "sampling/sampling_logp_difference/max": 1.5906305313110352, "sampling/importance_sampling_ratio/min": 0.20379707217216492, "sampling/importance_sampling_ratio/mean": 1.0070772171020508, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6710438020527363, "clip_ratio/low_mean": 0.04657141771167517, "clip_ratio/low_min": 0.04657141771167517, "clip_ratio/high_mean": 0.05450416449457407, "clip_ratio/high_max": 0.05450416449457407, "clip_ratio/region_mean": 0.10107558220624924, "reward_total_mean": 0.6534411907196045, "reward_meter_mean": 0.6779184341430664, "reward_meter_std": 0.355674147605896, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9150282740592957, "reward_repeat_soft_std": 0.05642801895737648, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.6534411907196045, "reward_total_composite_std": 0.15114767849445343} {"timestamp_utc": "2026-04-13T02:35:17Z", "mode": "eval", "global_step": 1700, "epoch": 0.17076845806127575, "eval_loss": NaN, "eval_runtime": 51.811, "eval_samples_per_second": 1.544, "eval_steps_per_second": 0.193, "eval_num_tokens": 3076495.0, "eval_completions/mean_length": 93.275, "eval_completions/min_length": 40.2, "eval_completions/max_length": 187.4, "eval_completions/clipped_ratio": 0.0125, "eval_completions/mean_terminated_length": 87.41250038146973, "eval_completions/min_terminated_length": 40.2, "eval_completions/max_terminated_length": 144.2, "eval_rewards/meter/mean": 0.7797888934612274, "eval_rewards/meter/std": 0.2890812013298273, "eval_rewards/count_adherence/mean": 0.9741666674613952, "eval_rewards/count_adherence/std": 0.049388696625828746, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.046291005611419675, "eval_rewards/repeat_soft/mean": 0.882794314622879, "eval_rewards/repeat_soft/std": 0.1029785305261612, "eval_rewards/judge_quality/mean": 0.357874995470047, "eval_rewards/judge_quality/std": 0.10915669873356819, "eval_rewards/total_composite/mean": 0.682727438211441, "eval_rewards/total_composite/std": 0.15023643970489503, "eval_reward": 0.682727438211441, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.054890270531177524, "eval_sampling/sampling_logp_difference/max": 0.9968271851539612, "eval_sampling/importance_sampling_ratio/min": 0.3771014630794525, "eval_sampling/importance_sampling_ratio/mean": 1.0121851444244385, "eval_sampling/importance_sampling_ratio/max": 1.4324937820434571, "eval_entropy": 0.5759748131036758, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.682727438211441, "eval_reward_meter_mean": 0.7797888934612274, "eval_reward_meter_std": 0.2890812013298273, "eval_reward_count_adherence_mean": 0.9741666674613952, "eval_reward_count_adherence_std": 0.049388696625828746, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.046291005611419675, "eval_reward_repeat_soft_mean": 0.882794314622879, "eval_reward_repeat_soft_std": 0.1029785305261612, "eval_reward_judge_quality_mean": 0.357874995470047, "eval_reward_judge_quality_std": 0.10915669873356819, "eval_reward_total_composite_mean": 0.682727438211441, "eval_reward_total_composite_std": 0.15023643970489503} {"timestamp_utc": "2026-04-13T02:35:28Z", "mode": "train", "global_step": 1701, "epoch": 0.17086891009542943, "loss": 0.0506, "grad_norm": 7.862242698669434, "learning_rate": 4.848484848484849e-06, "num_tokens": 3078618.0, "completions/mean_length": 97.375, "completions/min_length": 87.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.375, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.9805243015289307, "rewards/meter/std": 0.00787384994328022, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8321793079376221, "rewards/repeat_soft/std": 0.0753021165728569, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.771203875541687, "rewards/total_composite/std": 0.03328252583742142, "reward": 0.771203875541687, "reward_std": 0.03328252583742142, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0948963388800621, "sampling/sampling_logp_difference/max": 2.0163207054138184, "sampling/importance_sampling_ratio/min": 0.13314445316791534, "sampling/importance_sampling_ratio/mean": 1.004865288734436, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5058380328118801, "clip_ratio/low_mean": 0.045654647052288055, "clip_ratio/low_min": 0.045654647052288055, "clip_ratio/high_mean": 0.04145143600180745, "clip_ratio/high_max": 0.04145143600180745, "clip_ratio/region_mean": 0.0871060830540955, "reward_total_mean": 0.771203875541687, "reward_meter_mean": 0.9805243015289307, "reward_meter_std": 0.00787384994328022, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8321793079376221, "reward_repeat_soft_std": 0.0753021165728569, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.771203875541687, "reward_total_composite_std": 0.03328252583742142} {"timestamp_utc": "2026-04-13T02:35:36Z", "mode": "train", "global_step": 1702, "epoch": 0.17096936212958314, "loss": 0.0208, "grad_norm": 5.127974510192871, "learning_rate": 4.845454545454546e-06, "num_tokens": 3080951.0, "completions/mean_length": 122.625, "completions/min_length": 117.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.625, "completions/min_terminated_length": 117.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.9936942458152771, "rewards/meter/std": 0.002224273979663849, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8797354102134705, "rewards/repeat_soft/std": 0.07440874725580215, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.7818859815597534, "rewards/total_composite/std": 0.03373588249087334, "reward": 0.7818859815597534, "reward_std": 0.03373587876558304, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0807253047823906, "sampling/sampling_logp_difference/max": 2.2689857482910156, "sampling/importance_sampling_ratio/min": 0.10341701656579971, "sampling/importance_sampling_ratio/mean": 1.0024679899215698, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4095946438610554, "clip_ratio/low_mean": 0.03792042704299092, "clip_ratio/low_min": 0.03792042704299092, "clip_ratio/high_mean": 0.04291223827749491, "clip_ratio/high_max": 0.04291223827749491, "clip_ratio/region_mean": 0.08083266532048583, "reward_total_mean": 0.7818859815597534, "reward_meter_mean": 0.9936942458152771, "reward_meter_std": 0.002224273979663849, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8797354102134705, "reward_repeat_soft_std": 0.07440874725580215, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.7818859815597534, "reward_total_composite_std": 0.03373588249087334} {"timestamp_utc": "2026-04-13T02:35:44Z", "mode": "train", "global_step": 1703, "epoch": 0.17106981416373682, "loss": 0.0234, "grad_norm": 7.221065044403076, "learning_rate": 4.842424242424243e-06, "num_tokens": 3083026.0, "completions/mean_length": 84.375, "completions/min_length": 75.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.375, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9067833423614502, "rewards/meter/std": 0.16012556850910187, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9390347599983215, "rewards/repeat_soft/std": 0.043719854205846786, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7588309049606323, "rewards/total_composite/std": 0.0828794538974762, "reward": 0.7588309049606323, "reward_std": 0.0828794464468956, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0936276838183403, "sampling/sampling_logp_difference/max": 2.292764186859131, "sampling/importance_sampling_ratio/min": 0.10098693519830704, "sampling/importance_sampling_ratio/mean": 1.0086098909378052, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5408608205616474, "clip_ratio/low_mean": 0.0235294122248888, "clip_ratio/low_min": 0.0235294122248888, "clip_ratio/high_mean": 0.0779268010519445, "clip_ratio/high_max": 0.0779268010519445, "clip_ratio/region_mean": 0.1014562132768333, "reward_total_mean": 0.7588309049606323, "reward_meter_mean": 0.9067833423614502, "reward_meter_std": 0.16012556850910187, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9390347599983215, "reward_repeat_soft_std": 0.043719854205846786, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7588309049606323, "reward_total_composite_std": 0.0828794538974762} {"timestamp_utc": "2026-04-13T02:35:51Z", "mode": "train", "global_step": 1704, "epoch": 0.1711702661978905, "loss": 0.0365, "grad_norm": 9.450439453125, "learning_rate": 4.83939393939394e-06, "num_tokens": 3084752.0, "completions/mean_length": 46.75, "completions/min_length": 40.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.75, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9010294675827026, "rewards/meter/std": 0.1643972396850586, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9657539129257202, "rewards/repeat_soft/std": 0.0409851111471653, "rewards/judge_quality/mean": 0.7362500429153442, "rewards/judge_quality/std": 0.25376805663108826, "rewards/total_composite/mean": 0.872913658618927, "rewards/total_composite/std": 0.10068617016077042, "reward": 0.872913658618927, "reward_std": 0.10068617016077042, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12463430315256119, "sampling/sampling_logp_difference/max": 1.0894074440002441, "sampling/importance_sampling_ratio/min": 0.33641576766967773, "sampling/importance_sampling_ratio/mean": 1.0094877481460571, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6771477907896042, "clip_ratio/low_mean": 0.05384656507521868, "clip_ratio/low_min": 0.05384656507521868, "clip_ratio/high_mean": 0.05667334794998169, "clip_ratio/high_max": 0.05667334794998169, "clip_ratio/region_mean": 0.11051991302520037, "reward_total_mean": 0.872913658618927, "reward_meter_mean": 0.9010294675827026, "reward_meter_std": 0.1643972396850586, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9657539129257202, "reward_repeat_soft_std": 0.0409851111471653, "reward_judge_quality_mean": 0.7362500429153442, "reward_judge_quality_std": 0.25376805663108826, "reward_total_composite_mean": 0.872913658618927, "reward_total_composite_std": 0.10068617016077042} {"timestamp_utc": "2026-04-13T02:35:57Z", "mode": "train", "global_step": 1705, "epoch": 0.1712707182320442, "loss": 0.0054, "grad_norm": 21.675151824951172, "learning_rate": 4.836363636363637e-06, "num_tokens": 3086282.0, "completions/mean_length": 29.25, "completions/min_length": 24.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.25, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.44524407386779785, "rewards/meter/std": 0.4352852702140808, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.958458423614502, "rewards/repeat_soft/std": 0.011431216262280941, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404788017273, "rewards/total_composite/mean": 0.6130806803703308, "rewards/total_composite/std": 0.15909980237483978, "reward": 0.6130806803703308, "reward_std": 0.15909980237483978, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.162113755941391, "sampling/sampling_logp_difference/max": 1.5925034284591675, "sampling/importance_sampling_ratio/min": 0.20341572165489197, "sampling/importance_sampling_ratio/mean": 0.9893162250518799, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.033596359193325, "clip_ratio/low_mean": 0.06268252991139889, "clip_ratio/low_min": 0.06268252991139889, "clip_ratio/high_mean": 0.05737179610878229, "clip_ratio/high_max": 0.05737179610878229, "clip_ratio/region_mean": 0.12005432602018118, "reward_total_mean": 0.6130806803703308, "reward_meter_mean": 0.44524407386779785, "reward_meter_std": 0.4352852702140808, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.958458423614502, "reward_repeat_soft_std": 0.011431216262280941, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404788017273, "reward_total_composite_mean": 0.6130806803703308, "reward_total_composite_std": 0.15909980237483978} {"timestamp_utc": "2026-04-13T02:36:04Z", "mode": "train", "global_step": 1706, "epoch": 0.1713711702661979, "loss": 0.0663, "grad_norm": 13.015522003173828, "learning_rate": 4.833333333333333e-06, "num_tokens": 3087921.0, "completions/mean_length": 56.875, "completions/min_length": 48.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.875, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.7992294430732727, "rewards/meter/std": 0.3221721649169922, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9571216702461243, "rewards/repeat_soft/std": 0.03376096487045288, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7313653826713562, "rewards/total_composite/std": 0.1428014636039734, "reward": 0.7313653826713562, "reward_std": 0.1428014636039734, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12384092807769775, "sampling/sampling_logp_difference/max": 1.696213722229004, "sampling/importance_sampling_ratio/min": 0.1833765208721161, "sampling/importance_sampling_ratio/mean": 1.0035845041275024, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7279503643512726, "clip_ratio/low_mean": 0.03628095984458923, "clip_ratio/low_min": 0.03628095984458923, "clip_ratio/high_mean": 0.09151252871379256, "clip_ratio/high_max": 0.09151252871379256, "clip_ratio/region_mean": 0.1277934885583818, "reward_total_mean": 0.7313653826713562, "reward_meter_mean": 0.7992294430732727, "reward_meter_std": 0.3221721649169922, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9571216702461243, "reward_repeat_soft_std": 0.03376096487045288, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7313653826713562, "reward_total_composite_std": 0.1428014636039734} {"timestamp_utc": "2026-04-13T02:36:11Z", "mode": "train", "global_step": 1707, "epoch": 0.1714716223003516, "loss": 0.0159, "grad_norm": 10.237049102783203, "learning_rate": 4.830303030303031e-06, "num_tokens": 3089653.0, "completions/mean_length": 54.5, "completions/min_length": 46.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.5, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.7554166316986084, "rewards/meter/std": 0.32983285188674927, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9700720310211182, "rewards/repeat_soft/std": 0.02582109533250332, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.732819676399231, "rewards/total_composite/std": 0.16935104131698608, "reward": 0.732819676399231, "reward_std": 0.1693510264158249, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1177181676030159, "sampling/sampling_logp_difference/max": 1.1550016403198242, "sampling/importance_sampling_ratio/min": 0.31505700945854187, "sampling/importance_sampling_ratio/mean": 0.9974896907806396, "sampling/importance_sampling_ratio/max": 1.8120651245117188, "entropy": 0.6806580945849419, "clip_ratio/low_mean": 0.03503546956926584, "clip_ratio/low_min": 0.03503546956926584, "clip_ratio/high_mean": 0.07323061488568783, "clip_ratio/high_max": 0.07323061488568783, "clip_ratio/region_mean": 0.10826608445495367, "reward_total_mean": 0.732819676399231, "reward_meter_mean": 0.7554166316986084, "reward_meter_std": 0.32983285188674927, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9700720310211182, "reward_repeat_soft_std": 0.02582109533250332, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.732819676399231, "reward_total_composite_std": 0.16935104131698608} {"timestamp_utc": "2026-04-13T02:36:17Z", "mode": "train", "global_step": 1708, "epoch": 0.17157207433450528, "loss": 0.0436, "grad_norm": 13.704949378967285, "learning_rate": 4.827272727272728e-06, "num_tokens": 3091391.0, "completions/mean_length": 48.25, "completions/min_length": 43.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.25, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9739916920661926, "rewards/meter/std": 0.02009562961757183, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9543459415435791, "rewards/repeat_soft/std": 0.03365955874323845, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.8078558444976807, "rewards/total_composite/std": 0.031026966869831085, "reward": 0.8078558444976807, "reward_std": 0.031026989221572876, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09023729711771011, "sampling/sampling_logp_difference/max": 1.5717406272888184, "sampling/importance_sampling_ratio/min": 0.20768336951732635, "sampling/importance_sampling_ratio/mean": 1.0029716491699219, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4692523181438446, "clip_ratio/low_mean": 0.012019230984151363, "clip_ratio/low_min": 0.012019230984151363, "clip_ratio/high_mean": 0.08625076524913311, "clip_ratio/high_max": 0.08625076524913311, "clip_ratio/region_mean": 0.09826999623328447, "reward_total_mean": 0.8078558444976807, "reward_meter_mean": 0.9739916920661926, "reward_meter_std": 0.02009562961757183, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9543459415435791, "reward_repeat_soft_std": 0.03365955874323845, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.8078558444976807, "reward_total_composite_std": 0.031026966869831085} {"timestamp_utc": "2026-04-13T02:36:24Z", "mode": "train", "global_step": 1709, "epoch": 0.17167252636865896, "loss": 0.0494, "grad_norm": 11.467790603637695, "learning_rate": 4.824242424242424e-06, "num_tokens": 3093114.0, "completions/mean_length": 49.375, "completions/min_length": 44.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.375, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9603428244590759, "rewards/meter/std": 0.06581888347864151, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9719963073730469, "rewards/repeat_soft/std": 0.04003526270389557, "rewards/judge_quality/mean": 0.39249998331069946, "rewards/judge_quality/std": 0.08892211318016052, "rewards/total_composite/mean": 0.7971038818359375, "rewards/total_composite/std": 0.039940088987350464, "reward": 0.7971038818359375, "reward_std": 0.039940085262060165, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10026401281356812, "sampling/sampling_logp_difference/max": 2.20159649848938, "sampling/importance_sampling_ratio/min": 0.11062640696763992, "sampling/importance_sampling_ratio/mean": 1.0160075426101685, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5801496282219887, "clip_ratio/low_mean": 0.05182987917214632, "clip_ratio/low_min": 0.05182987917214632, "clip_ratio/high_mean": 0.049241832457482815, "clip_ratio/high_max": 0.049241832457482815, "clip_ratio/region_mean": 0.10107171162962914, "reward_total_mean": 0.7971038818359375, "reward_meter_mean": 0.9603428244590759, "reward_meter_std": 0.06581888347864151, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9719963073730469, "reward_repeat_soft_std": 0.04003526270389557, "reward_judge_quality_mean": 0.39249998331069946, "reward_judge_quality_std": 0.08892211318016052, "reward_total_composite_mean": 0.7971038818359375, "reward_total_composite_std": 0.039940088987350464} {"timestamp_utc": "2026-04-13T02:36:30Z", "mode": "train", "global_step": 1710, "epoch": 0.17177297840281266, "loss": 0.0245, "grad_norm": 15.5913724899292, "learning_rate": 4.8212121212121215e-06, "num_tokens": 3095105.0, "completions/mean_length": 80.875, "completions/min_length": 73.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.875, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9461846947669983, "rewards/meter/std": 0.08904185146093369, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8822543025016785, "rewards/repeat_soft/std": 0.10642741620540619, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.1527603268623352, "rewards/total_composite/mean": 0.7922585010528564, "rewards/total_composite/std": 0.07923062890768051, "reward": 0.7922585010528564, "reward_std": 0.07923060655593872, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12923187017440796, "sampling/sampling_logp_difference/max": 1.5916213989257812, "sampling/importance_sampling_ratio/min": 0.20359523594379425, "sampling/importance_sampling_ratio/mean": 1.038291096687317, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8499169908463955, "clip_ratio/low_mean": 0.03975159954279661, "clip_ratio/low_min": 0.03975159954279661, "clip_ratio/high_mean": 0.0701736519113183, "clip_ratio/high_max": 0.0701736519113183, "clip_ratio/region_mean": 0.10992525145411491, "reward_total_mean": 0.7922585010528564, "reward_meter_mean": 0.9461846947669983, "reward_meter_std": 0.08904185146093369, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8822543025016785, "reward_repeat_soft_std": 0.10642741620540619, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.1527603268623352, "reward_total_composite_mean": 0.7922585010528564, "reward_total_composite_std": 0.07923062890768051} {"timestamp_utc": "2026-04-13T02:36:38Z", "mode": "train", "global_step": 1711, "epoch": 0.17187343043696635, "loss": 0.0462, "grad_norm": 6.030894756317139, "learning_rate": 4.818181818181819e-06, "num_tokens": 3097950.0, "completions/mean_length": 170.625, "completions/min_length": 150.0, "completions/max_length": 187.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 170.625, "completions/min_terminated_length": 150.0, "completions/max_terminated_length": 187.0, "rewards/meter/mean": 0.5822442173957825, "rewards/meter/std": 0.3227042555809021, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.907263457775116, "rewards/repeat_soft/std": 0.07264033704996109, "rewards/judge_quality/mean": 0.2762500047683716, "rewards/judge_quality/std": 0.12603145837783813, "rewards/total_composite/mean": 0.5856112241744995, "rewards/total_composite/std": 0.1423042267560959, "reward": 0.5856112241744995, "reward_std": 0.1423042118549347, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10886459052562714, "sampling/sampling_logp_difference/max": 2.191587448120117, "sampling/importance_sampling_ratio/min": 0.11173922568559647, "sampling/importance_sampling_ratio/mean": 1.0131882429122925, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6535704135894775, "clip_ratio/low_mean": 0.05409695953130722, "clip_ratio/low_min": 0.05409695953130722, "clip_ratio/high_mean": 0.028788375668227673, "clip_ratio/high_max": 0.028788375668227673, "clip_ratio/region_mean": 0.08288533519953489, "reward_total_mean": 0.5856112241744995, "reward_meter_mean": 0.5822442173957825, "reward_meter_std": 0.3227042555809021, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.907263457775116, "reward_repeat_soft_std": 0.07264033704996109, "reward_judge_quality_mean": 0.2762500047683716, "reward_judge_quality_std": 0.12603145837783813, "reward_total_composite_mean": 0.5856112241744995, "reward_total_composite_std": 0.1423042267560959} {"timestamp_utc": "2026-04-13T02:36:45Z", "mode": "train", "global_step": 1712, "epoch": 0.17197388247112003, "loss": -0.0088, "grad_norm": 10.949590682983398, "learning_rate": 4.815151515151515e-06, "num_tokens": 3099675.0, "completions/mean_length": 51.625, "completions/min_length": 29.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.625, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.7145391702651978, "rewards/meter/std": 0.37130007147789, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9564615488052368, "rewards/repeat_soft/std": 0.06065477430820465, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.28263744711875916, "rewards/total_composite/mean": 0.7059387564659119, "rewards/total_composite/std": 0.18209129571914673, "reward": 0.7059387564659119, "reward_std": 0.18209131062030792, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1264832615852356, "sampling/sampling_logp_difference/max": 1.6047697067260742, "sampling/importance_sampling_ratio/min": 0.20093581080436707, "sampling/importance_sampling_ratio/mean": 1.0108435153961182, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7107297629117966, "clip_ratio/low_mean": 0.03647926356643438, "clip_ratio/low_min": 0.03647926356643438, "clip_ratio/high_mean": 0.086586132645607, "clip_ratio/high_max": 0.086586132645607, "clip_ratio/region_mean": 0.12306539621204138, "reward_total_mean": 0.7059387564659119, "reward_meter_mean": 0.7145391702651978, "reward_meter_std": 0.37130007147789, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9564615488052368, "reward_repeat_soft_std": 0.06065477430820465, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.28263744711875916, "reward_total_composite_mean": 0.7059387564659119, "reward_total_composite_std": 0.18209129571914673} {"timestamp_utc": "2026-04-13T02:36:57Z", "mode": "train", "global_step": 1713, "epoch": 0.17207433450527373, "loss": -0.1985, "grad_norm": 1.9965026378631592, "learning_rate": 4.8121212121212125e-06, "num_tokens": 3101649.0, "completions/mean_length": 151.75, "completions/min_length": 84.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 100.28572082519531, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.9296243190765381, "rewards/meter/std": 0.11123079806566238, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.26726123690605164, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9316638708114624, "rewards/repeat_soft/std": 0.05508211627602577, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.13715866208076477, "rewards/total_composite/mean": 0.6677684187889099, "rewards/total_composite/std": 0.2774823307991028, "reward": 0.6677684187889099, "reward_std": 0.2774823009967804, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11312650889158249, "sampling/sampling_logp_difference/max": 1.7975974082946777, "sampling/importance_sampling_ratio/min": 0.16569651663303375, "sampling/importance_sampling_ratio/mean": 1.0119253396987915, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5537709817290306, "clip_ratio/low_mean": 0.010416666977107525, "clip_ratio/low_min": 0.010416666977107525, "clip_ratio/high_mean": 0.09547118470072746, "clip_ratio/high_max": 0.09547118470072746, "clip_ratio/region_mean": 0.10588785167783499, "reward_total_mean": 0.6677684187889099, "reward_meter_mean": 0.9296243190765381, "reward_meter_std": 0.11123079806566238, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.26726123690605164, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9316638708114624, "reward_repeat_soft_std": 0.05508211627602577, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.13715866208076477, "reward_total_composite_mean": 0.6677684187889099, "reward_total_composite_std": 0.2774823307991028} {"timestamp_utc": "2026-04-13T02:37:09Z", "mode": "train", "global_step": 1714, "epoch": 0.17217478653942742, "loss": -0.1094, "grad_norm": 1.97781240940094, "learning_rate": 4.80909090909091e-06, "num_tokens": 3103097.0, "completions/mean_length": 96.0, "completions/min_length": 33.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 36.57143020629883, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.7563886642456055, "rewards/meter/std": 0.3422507643699646, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.986402153968811, "rewards/repeat_soft/std": 0.017342062667012215, "rewards/judge_quality/mean": 0.39249998331069946, "rewards/judge_quality/std": 0.13905291259288788, "rewards/total_composite/mean": 0.6736390590667725, "rewards/total_composite/std": 0.2815326750278473, "reward": 0.6736390590667725, "reward_std": 0.2815326750278473, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12681567668914795, "sampling/sampling_logp_difference/max": 1.5601608753204346, "sampling/importance_sampling_ratio/min": 0.3974047303199768, "sampling/importance_sampling_ratio/mean": 1.0360569953918457, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7050383239984512, "clip_ratio/low_mean": 0.012820512987673283, "clip_ratio/low_min": 0.012820512987673283, "clip_ratio/high_mean": 0.0842931205406785, "clip_ratio/high_max": 0.0842931205406785, "clip_ratio/region_mean": 0.09711363352835178, "reward_total_mean": 0.6736390590667725, "reward_meter_mean": 0.7563886642456055, "reward_meter_std": 0.3422507643699646, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.986402153968811, "reward_repeat_soft_std": 0.017342062667012215, "reward_judge_quality_mean": 0.39249998331069946, "reward_judge_quality_std": 0.13905291259288788, "reward_total_composite_mean": 0.6736390590667725, "reward_total_composite_std": 0.2815326750278473} {"timestamp_utc": "2026-04-13T02:37:21Z", "mode": "train", "global_step": 1715, "epoch": 0.17227523857358112, "loss": -0.1516, "grad_norm": 1.8568528890609741, "learning_rate": 4.806060606060606e-06, "num_tokens": 3104860.0, "completions/mean_length": 114.375, "completions/min_length": 55.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 57.57143020629883, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9872651100158691, "rewards/meter/std": 0.007564142346382141, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9484661817550659, "rewards/repeat_soft/std": 0.06496556103229523, "rewards/judge_quality/mean": 0.6187499761581421, "rewards/judge_quality/std": 0.2909805178642273, "rewards/total_composite/mean": 0.785963773727417, "rewards/total_composite/std": 0.3226819932460785, "reward": 0.785963773727417, "reward_std": 0.3226819634437561, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11959080398082733, "sampling/sampling_logp_difference/max": 1.6220124959945679, "sampling/importance_sampling_ratio/min": 0.1975008249282837, "sampling/importance_sampling_ratio/mean": 1.0269932746887207, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.640937015414238, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09962496720254421, "clip_ratio/high_max": 0.09962496720254421, "clip_ratio/region_mean": 0.09962496720254421, "reward_total_mean": 0.785963773727417, "reward_meter_mean": 0.9872651100158691, "reward_meter_std": 0.007564142346382141, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9484661817550659, "reward_repeat_soft_std": 0.06496556103229523, "reward_judge_quality_mean": 0.6187499761581421, "reward_judge_quality_std": 0.2909805178642273, "reward_total_composite_mean": 0.785963773727417, "reward_total_composite_std": 0.3226819932460785} {"timestamp_utc": "2026-04-13T02:37:31Z", "mode": "train", "global_step": 1716, "epoch": 0.1723756906077348, "loss": 0.0289, "grad_norm": 12.655372619628906, "learning_rate": 4.803030303030303e-06, "num_tokens": 3106643.0, "completions/mean_length": 51.875, "completions/min_length": 47.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.875, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.7803674936294556, "rewards/meter/std": 0.3454989790916443, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9760687351226807, "rewards/repeat_soft/std": 0.017330707982182503, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7258971929550171, "rewards/total_composite/std": 0.15537050366401672, "reward": 0.7258971929550171, "reward_std": 0.15537048876285553, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13123153150081635, "sampling/sampling_logp_difference/max": 1.98565673828125, "sampling/importance_sampling_ratio/min": 0.13729041814804077, "sampling/importance_sampling_ratio/mean": 1.0164265632629395, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9265305623412132, "clip_ratio/low_mean": 0.026279956102371216, "clip_ratio/low_min": 0.026279956102371216, "clip_ratio/high_mean": 0.08407053910195827, "clip_ratio/high_max": 0.08407053910195827, "clip_ratio/region_mean": 0.11035049520432949, "reward_total_mean": 0.7258971929550171, "reward_meter_mean": 0.7803674936294556, "reward_meter_std": 0.3454989790916443, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9760687351226807, "reward_repeat_soft_std": 0.017330707982182503, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7258971929550171, "reward_total_composite_std": 0.15537050366401672} {"timestamp_utc": "2026-04-13T02:37:38Z", "mode": "train", "global_step": 1717, "epoch": 0.17247614264188849, "loss": 0.058, "grad_norm": 12.121832847595215, "learning_rate": 4.800000000000001e-06, "num_tokens": 3108331.0, "completions/mean_length": 41.0, "completions/min_length": 35.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.6520545482635498, "rewards/meter/std": 0.35838013887405396, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9847122430801392, "rewards/repeat_soft/std": 0.012998953461647034, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.7616457939147949, "rewards/total_composite/std": 0.18007270991802216, "reward": 0.7616457939147949, "reward_std": 0.18007269501686096, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14369933307170868, "sampling/sampling_logp_difference/max": 1.7931513786315918, "sampling/importance_sampling_ratio/min": 0.16643483936786652, "sampling/importance_sampling_ratio/mean": 0.996990442276001, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6249169409275055, "clip_ratio/low_mean": 0.058961291797459126, "clip_ratio/low_min": 0.058961291797459126, "clip_ratio/high_mean": 0.06994831003248692, "clip_ratio/high_max": 0.06994831003248692, "clip_ratio/region_mean": 0.12890960182994604, "reward_total_mean": 0.7616457939147949, "reward_meter_mean": 0.6520545482635498, "reward_meter_std": 0.35838013887405396, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9847122430801392, "reward_repeat_soft_std": 0.012998953461647034, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.7616457939147949, "reward_total_composite_std": 0.18007270991802216} {"timestamp_utc": "2026-04-13T02:37:44Z", "mode": "train", "global_step": 1718, "epoch": 0.1725765946760422, "loss": 0.0187, "grad_norm": 19.636178970336914, "learning_rate": 4.796969696969697e-06, "num_tokens": 3109916.0, "completions/mean_length": 26.125, "completions/min_length": 22.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.125, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.8676780462265015, "rewards/meter/std": 0.20960238575935364, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9622748494148254, "rewards/repeat_soft/std": 0.0006367546156980097, "rewards/judge_quality/mean": 0.6187499761581421, "rewards/judge_quality/std": 0.24976776540279388, "rewards/total_composite/mean": 0.8223075866699219, "rewards/total_composite/std": 0.07830572128295898, "reward": 0.8223075866699219, "reward_std": 0.07830573618412018, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15437349677085876, "sampling/sampling_logp_difference/max": 1.1322441101074219, "sampling/importance_sampling_ratio/min": 0.32230913639068604, "sampling/importance_sampling_ratio/mean": 1.0381810665130615, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.851251132786274, "clip_ratio/low_mean": 0.04734169598668814, "clip_ratio/low_min": 0.04734169598668814, "clip_ratio/high_mean": 0.04297911189496517, "clip_ratio/high_max": 0.04297911189496517, "clip_ratio/region_mean": 0.09032080788165331, "reward_total_mean": 0.8223075866699219, "reward_meter_mean": 0.8676780462265015, "reward_meter_std": 0.20960238575935364, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9622748494148254, "reward_repeat_soft_std": 0.0006367546156980097, "reward_judge_quality_mean": 0.6187499761581421, "reward_judge_quality_std": 0.24976776540279388, "reward_total_composite_mean": 0.8223075866699219, "reward_total_composite_std": 0.07830572128295898} {"timestamp_utc": "2026-04-13T02:37:51Z", "mode": "train", "global_step": 1719, "epoch": 0.17267704671019588, "loss": 0.0152, "grad_norm": 10.437137603759766, "learning_rate": 4.793939393939394e-06, "num_tokens": 3111687.0, "completions/mean_length": 58.375, "completions/min_length": 48.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.375, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.8293755054473877, "rewards/meter/std": 0.2196078598499298, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9646583795547485, "rewards/repeat_soft/std": 0.024424709379673004, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.783184826374054, "rewards/total_composite/std": 0.14089497923851013, "reward": 0.783184826374054, "reward_std": 0.14089496433734894, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14043235778808594, "sampling/sampling_logp_difference/max": 1.1698276996612549, "sampling/importance_sampling_ratio/min": 0.310420423746109, "sampling/importance_sampling_ratio/mean": 1.026005506515503, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9568302929401398, "clip_ratio/low_mean": 0.049691442400217056, "clip_ratio/low_min": 0.049691442400217056, "clip_ratio/high_mean": 0.08831292483955622, "clip_ratio/high_max": 0.08831292483955622, "clip_ratio/region_mean": 0.13800436723977327, "reward_total_mean": 0.783184826374054, "reward_meter_mean": 0.8293755054473877, "reward_meter_std": 0.2196078598499298, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9646583795547485, "reward_repeat_soft_std": 0.024424709379673004, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.783184826374054, "reward_total_composite_std": 0.14089497923851013} {"timestamp_utc": "2026-04-13T02:37:58Z", "mode": "train", "global_step": 1720, "epoch": 0.17277749874434958, "loss": 0.04, "grad_norm": 9.642221450805664, "learning_rate": 4.790909090909091e-06, "num_tokens": 3113992.0, "completions/mean_length": 108.125, "completions/min_length": 102.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.125, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.9591643810272217, "rewards/meter/std": 0.07928570359945297, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.928742527961731, "rewards/repeat_soft/std": 0.04844045266509056, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.8145607113838196, "rewards/total_composite/std": 0.06571466475725174, "reward": 0.8145607113838196, "reward_std": 0.06571465730667114, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11224285513162613, "sampling/sampling_logp_difference/max": 1.5249881744384766, "sampling/importance_sampling_ratio/min": 0.21762363612651825, "sampling/importance_sampling_ratio/mean": 1.0060865879058838, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6790691688656807, "clip_ratio/low_mean": 0.050462753511965275, "clip_ratio/low_min": 0.050462753511965275, "clip_ratio/high_mean": 0.04375407937914133, "clip_ratio/high_max": 0.04375407937914133, "clip_ratio/region_mean": 0.0942168328911066, "reward_total_mean": 0.8145607113838196, "reward_meter_mean": 0.9591643810272217, "reward_meter_std": 0.07928570359945297, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.928742527961731, "reward_repeat_soft_std": 0.04844045266509056, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.8145607113838196, "reward_total_composite_std": 0.06571466475725174} {"timestamp_utc": "2026-04-13T02:38:10Z", "mode": "train", "global_step": 1721, "epoch": 0.17287795077850326, "loss": -0.126, "grad_norm": 1.6057120561599731, "learning_rate": 4.787878787878788e-06, "num_tokens": 3115620.0, "completions/mean_length": 168.5, "completions/min_length": 51.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.7228949069976807, "rewards/meter/std": 0.3837904930114746, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.3720119297504425, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9888018369674683, "rewards/repeat_soft/std": 0.014392617158591747, "rewards/judge_quality/mean": 0.45625001192092896, "rewards/judge_quality/std": 0.3304083049297333, "rewards/total_composite/mean": 0.6289299130439758, "rewards/total_composite/std": 0.39068734645843506, "reward": 0.6289299130439758, "reward_std": 0.39068734645843506, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11852795630693436, "sampling/sampling_logp_difference/max": 0.9864153861999512, "sampling/importance_sampling_ratio/min": 0.3729110658168793, "sampling/importance_sampling_ratio/mean": 1.0188084840774536, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5987909808754921, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08869554242119193, "clip_ratio/high_max": 0.08869554242119193, "clip_ratio/region_mean": 0.08869554242119193, "reward_total_mean": 0.6289299130439758, "reward_meter_mean": 0.7228949069976807, "reward_meter_std": 0.3837904930114746, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.3720119297504425, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9888018369674683, "reward_repeat_soft_std": 0.014392617158591747, "reward_judge_quality_mean": 0.45625001192092896, "reward_judge_quality_std": 0.3304083049297333, "reward_total_composite_mean": 0.6289299130439758, "reward_total_composite_std": 0.39068734645843506} {"timestamp_utc": "2026-04-13T02:38:17Z", "mode": "train", "global_step": 1722, "epoch": 0.17297840281265695, "loss": 0.1172, "grad_norm": 21.81240463256836, "learning_rate": 4.784848484848485e-06, "num_tokens": 3117203.0, "completions/mean_length": 39.875, "completions/min_length": 32.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.875, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.7597067356109619, "rewards/meter/std": 0.30855968594551086, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9885257482528687, "rewards/repeat_soft/std": 0.021206922829151154, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.7917205691337585, "rewards/total_composite/std": 0.1675584316253662, "reward": 0.7917205691337585, "reward_std": 0.1675584316253662, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13426342606544495, "sampling/sampling_logp_difference/max": 1.6188616752624512, "sampling/importance_sampling_ratio/min": 0.19812411069869995, "sampling/importance_sampling_ratio/mean": 1.003769040107727, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.626220591366291, "clip_ratio/low_mean": 0.045242778956890106, "clip_ratio/low_min": 0.045242778956890106, "clip_ratio/high_mean": 0.06515230517834425, "clip_ratio/high_max": 0.06515230517834425, "clip_ratio/region_mean": 0.11039508413523436, "reward_total_mean": 0.7917205691337585, "reward_meter_mean": 0.7597067356109619, "reward_meter_std": 0.30855968594551086, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9885257482528687, "reward_repeat_soft_std": 0.021206922829151154, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.7917205691337585, "reward_total_composite_std": 0.1675584316253662} {"timestamp_utc": "2026-04-13T02:38:29Z", "mode": "train", "global_step": 1723, "epoch": 0.17307885484681065, "loss": -0.0966, "grad_norm": 3.319488525390625, "learning_rate": 4.7818181818181825e-06, "num_tokens": 3118734.0, "completions/mean_length": 101.375, "completions/min_length": 39.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 42.71428680419922, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.7431662082672119, "rewards/meter/std": 0.37025490403175354, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9856399297714233, "rewards/repeat_soft/std": 0.026714375242590904, "rewards/judge_quality/mean": 0.3675000071525574, "rewards/judge_quality/std": 0.14508618414402008, "rewards/total_composite/mean": 0.616675853729248, "rewards/total_composite/std": 0.2977468967437744, "reward": 0.616675853729248, "reward_std": 0.2977468967437744, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15056994557380676, "sampling/sampling_logp_difference/max": 2.487229108810425, "sampling/importance_sampling_ratio/min": 0.08314002305269241, "sampling/importance_sampling_ratio/mean": 0.9973805546760559, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7688239887356758, "clip_ratio/low_mean": 0.016025641933083534, "clip_ratio/low_min": 0.016025641933083534, "clip_ratio/high_mean": 0.0912700928747654, "clip_ratio/high_max": 0.0912700928747654, "clip_ratio/region_mean": 0.10729573480784893, "reward_total_mean": 0.616675853729248, "reward_meter_mean": 0.7431662082672119, "reward_meter_std": 0.37025490403175354, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9856399297714233, "reward_repeat_soft_std": 0.026714375242590904, "reward_judge_quality_mean": 0.3675000071525574, "reward_judge_quality_std": 0.14508618414402008, "reward_total_composite_mean": 0.616675853729248, "reward_total_composite_std": 0.2977468967437744} {"timestamp_utc": "2026-04-13T02:38:36Z", "mode": "train", "global_step": 1724, "epoch": 0.17317930688096433, "loss": 0.0764, "grad_norm": 15.367180824279785, "learning_rate": 4.77878787878788e-06, "num_tokens": 3120311.0, "completions/mean_length": 51.125, "completions/min_length": 44.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.125, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.5597729682922363, "rewards/meter/std": 0.3529880940914154, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9543853998184204, "rewards/repeat_soft/std": 0.07163773477077484, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.6795863509178162, "rewards/total_composite/std": 0.22394324839115143, "reward": 0.6795863509178162, "reward_std": 0.22394323348999023, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11813989281654358, "sampling/sampling_logp_difference/max": 1.076141357421875, "sampling/importance_sampling_ratio/min": 0.3409084379673004, "sampling/importance_sampling_ratio/mean": 0.9990083575248718, "sampling/importance_sampling_ratio/max": 1.9985524415969849, "entropy": 0.7492409497499466, "clip_ratio/low_mean": 0.07360413391143084, "clip_ratio/low_min": 0.07360413391143084, "clip_ratio/high_mean": 0.04666975885629654, "clip_ratio/high_max": 0.04666975885629654, "clip_ratio/region_mean": 0.12027389276772738, "reward_total_mean": 0.6795863509178162, "reward_meter_mean": 0.5597729682922363, "reward_meter_std": 0.3529880940914154, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9543853998184204, "reward_repeat_soft_std": 0.07163773477077484, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.6795863509178162, "reward_total_composite_std": 0.22394324839115143} {"timestamp_utc": "2026-04-13T02:38:42Z", "mode": "train", "global_step": 1725, "epoch": 0.17327975891511804, "loss": 0.0091, "grad_norm": 14.594118118286133, "learning_rate": 4.775757575757576e-06, "num_tokens": 3121960.0, "completions/mean_length": 38.125, "completions/min_length": 36.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.125, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.7242047786712646, "rewards/meter/std": 0.32868075370788574, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9651288986206055, "rewards/repeat_soft/std": 0.028553878888487816, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.7029050588607788, "rewards/total_composite/std": 0.14811356365680695, "reward": 0.7029050588607788, "reward_std": 0.14811354875564575, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10714028775691986, "sampling/sampling_logp_difference/max": 2.611720085144043, "sampling/importance_sampling_ratio/min": 0.07340817153453827, "sampling/importance_sampling_ratio/mean": 1.0057320594787598, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5121947526931763, "clip_ratio/low_mean": 0.034913004375994205, "clip_ratio/low_min": 0.034913004375994205, "clip_ratio/high_mean": 0.06540916091762483, "clip_ratio/high_max": 0.06540916091762483, "clip_ratio/region_mean": 0.10032216529361904, "reward_total_mean": 0.7029050588607788, "reward_meter_mean": 0.7242047786712646, "reward_meter_std": 0.32868075370788574, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9651288986206055, "reward_repeat_soft_std": 0.028553878888487816, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.7029050588607788, "reward_total_composite_std": 0.14811356365680695} {"timestamp_utc": "2026-04-13T02:38:49Z", "mode": "train", "global_step": 1726, "epoch": 0.17338021094927172, "loss": 0.084, "grad_norm": 8.584648132324219, "learning_rate": 4.772727272727273e-06, "num_tokens": 3124424.0, "completions/mean_length": 103.0, "completions/min_length": 96.0, "completions/max_length": 115.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.0, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 115.0, "rewards/meter/mean": 0.9560444951057434, "rewards/meter/std": 0.06665879487991333, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9222790598869324, "rewards/repeat_soft/std": 0.04469291493296623, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7920728921890259, "rewards/total_composite/std": 0.029239322990179062, "reward": 0.7920728921890259, "reward_std": 0.029239322990179062, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09351123124361038, "sampling/sampling_logp_difference/max": 1.7715835571289062, "sampling/importance_sampling_ratio/min": 0.17006348073482513, "sampling/importance_sampling_ratio/mean": 1.0146890878677368, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5416898652911186, "clip_ratio/low_mean": 0.025543478317558765, "clip_ratio/low_min": 0.025543478317558765, "clip_ratio/high_mean": 0.06899385340511799, "clip_ratio/high_max": 0.06899385340511799, "clip_ratio/region_mean": 0.09453733172267675, "reward_total_mean": 0.7920728921890259, "reward_meter_mean": 0.9560444951057434, "reward_meter_std": 0.06665879487991333, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9222790598869324, "reward_repeat_soft_std": 0.04469291493296623, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7920728921890259, "reward_total_composite_std": 0.029239322990179062} {"timestamp_utc": "2026-04-13T02:38:57Z", "mode": "train", "global_step": 1727, "epoch": 0.1734806629834254, "loss": 0.0252, "grad_norm": 7.6500372886657715, "learning_rate": 4.769696969696971e-06, "num_tokens": 3126548.0, "completions/mean_length": 107.5, "completions/min_length": 103.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.5, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.7549038529396057, "rewards/meter/std": 0.3172314465045929, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9369492530822754, "rewards/repeat_soft/std": 0.03925323113799095, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.19331690669059753, "rewards/total_composite/mean": 0.707901656627655, "rewards/total_composite/std": 0.1803063601255417, "reward": 0.707901656627655, "reward_std": 0.1803063601255417, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09769301861524582, "sampling/sampling_logp_difference/max": 1.965458869934082, "sampling/importance_sampling_ratio/min": 0.14009158313274384, "sampling/importance_sampling_ratio/mean": 1.007680892944336, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5634865835309029, "clip_ratio/low_mean": 0.03164758486673236, "clip_ratio/low_min": 0.03164758486673236, "clip_ratio/high_mean": 0.06013839412480593, "clip_ratio/high_max": 0.06013839412480593, "clip_ratio/region_mean": 0.09178597899153829, "reward_total_mean": 0.707901656627655, "reward_meter_mean": 0.7549038529396057, "reward_meter_std": 0.3172314465045929, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9369492530822754, "reward_repeat_soft_std": 0.03925323113799095, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.19331690669059753, "reward_total_composite_mean": 0.707901656627655, "reward_total_composite_std": 0.1803063601255417} {"timestamp_utc": "2026-04-13T02:39:05Z", "mode": "train", "global_step": 1728, "epoch": 0.1735811150175791, "loss": 0.0498, "grad_norm": 11.260504722595215, "learning_rate": 4.766666666666667e-06, "num_tokens": 3128378.0, "completions/mean_length": 52.75, "completions/min_length": 45.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.75, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.6315659880638123, "rewards/meter/std": 0.4208865463733673, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9919620752334595, "rewards/repeat_soft/std": 0.017439430579543114, "rewards/judge_quality/mean": 0.7150000333786011, "rewards/judge_quality/std": 0.24307554960250854, "rewards/total_composite/mean": 0.7479008436203003, "rewards/total_composite/std": 0.16810014843940735, "reward": 0.7479008436203003, "reward_std": 0.16810013353824615, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1315516233444214, "sampling/sampling_logp_difference/max": 1.4092903137207031, "sampling/importance_sampling_ratio/min": 0.24431660771369934, "sampling/importance_sampling_ratio/mean": 1.0126769542694092, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8725063875317574, "clip_ratio/low_mean": 0.04406565614044666, "clip_ratio/low_min": 0.04406565614044666, "clip_ratio/high_mean": 0.08541666902601719, "clip_ratio/high_max": 0.08541666902601719, "clip_ratio/region_mean": 0.12948232516646385, "reward_total_mean": 0.7479008436203003, "reward_meter_mean": 0.6315659880638123, "reward_meter_std": 0.4208865463733673, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9919620752334595, "reward_repeat_soft_std": 0.017439430579543114, "reward_judge_quality_mean": 0.7150000333786011, "reward_judge_quality_std": 0.24307554960250854, "reward_total_composite_mean": 0.7479008436203003, "reward_total_composite_std": 0.16810014843940735} {"timestamp_utc": "2026-04-13T02:39:12Z", "mode": "train", "global_step": 1729, "epoch": 0.1736815670517328, "loss": -0.0141, "grad_norm": 15.153559684753418, "learning_rate": 4.763636363636364e-06, "num_tokens": 3129785.0, "completions/mean_length": 24.875, "completions/min_length": 22.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.875, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9646182656288147, "rewards/meter/std": 0.04330988973379135, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9460431337356567, "rewards/repeat_soft/std": 0.04023751616477966, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8114325404167175, "rewards/total_composite/std": 0.023484492674469948, "reward": 0.8114325404167175, "reward_std": 0.023484481498599052, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14183661341667175, "sampling/sampling_logp_difference/max": 1.3894100189208984, "sampling/importance_sampling_ratio/min": 0.38469550013542175, "sampling/importance_sampling_ratio/mean": 1.0326290130615234, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9567270651459694, "clip_ratio/low_mean": 0.03903162106871605, "clip_ratio/low_min": 0.03903162106871605, "clip_ratio/high_mean": 0.09996553137898445, "clip_ratio/high_max": 0.09996553137898445, "clip_ratio/region_mean": 0.1389971524477005, "reward_total_mean": 0.8114325404167175, "reward_meter_mean": 0.9646182656288147, "reward_meter_std": 0.04330988973379135, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9460431337356567, "reward_repeat_soft_std": 0.04023751616477966, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8114325404167175, "reward_total_composite_std": 0.023484492674469948} {"timestamp_utc": "2026-04-13T02:39:19Z", "mode": "train", "global_step": 1730, "epoch": 0.1737820190858865, "loss": 0.0853, "grad_norm": 9.55591869354248, "learning_rate": 4.760606060606061e-06, "num_tokens": 3131521.0, "completions/mean_length": 51.0, "completions/min_length": 43.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.0, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.7340230941772461, "rewards/meter/std": 0.421207070350647, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9639407396316528, "rewards/repeat_soft/std": 0.031203847378492355, "rewards/judge_quality/mean": 0.4649999737739563, "rewards/judge_quality/std": 0.1940544992685318, "rewards/total_composite/mean": 0.7162044644355774, "rewards/total_composite/std": 0.20400801301002502, "reward": 0.7162044644355774, "reward_std": 0.20400801301002502, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13940778374671936, "sampling/sampling_logp_difference/max": 2.1762537956237793, "sampling/importance_sampling_ratio/min": 0.1134658008813858, "sampling/importance_sampling_ratio/mean": 0.9996318221092224, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7274619042873383, "clip_ratio/low_mean": 0.040070243179798126, "clip_ratio/low_min": 0.040070243179798126, "clip_ratio/high_mean": 0.12156882602721453, "clip_ratio/high_max": 0.12156882602721453, "clip_ratio/region_mean": 0.16163906920701265, "reward_total_mean": 0.7162044644355774, "reward_meter_mean": 0.7340230941772461, "reward_meter_std": 0.421207070350647, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9639407396316528, "reward_repeat_soft_std": 0.031203847378492355, "reward_judge_quality_mean": 0.4649999737739563, "reward_judge_quality_std": 0.1940544992685318, "reward_total_composite_mean": 0.7162044644355774, "reward_total_composite_std": 0.20400801301002502} {"timestamp_utc": "2026-04-13T02:39:32Z", "mode": "train", "global_step": 1731, "epoch": 0.17388247112004018, "loss": -0.2339, "grad_norm": 1.9917445182800293, "learning_rate": 4.757575757575758e-06, "num_tokens": 3134362.0, "completions/mean_length": 210.125, "completions/min_length": 135.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 167.0, "completions/min_terminated_length": 135.0, "completions/max_terminated_length": 183.0, "rewards/meter/mean": 0.7750146389007568, "rewards/meter/std": 0.2348131388425827, "rewards/count_adherence/mean": 0.8958333134651184, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8851284980773926, "rewards/repeat_soft/std": 0.09859985113143921, "rewards/judge_quality/mean": 0.2462500035762787, "rewards/judge_quality/std": 0.09913014620542526, "rewards/total_composite/mean": 0.5765939354896545, "rewards/total_composite/std": 0.2526305913925171, "reward": 0.5765939354896545, "reward_std": 0.2526305913925171, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09853669255971909, "sampling/sampling_logp_difference/max": 1.8535146713256836, "sampling/importance_sampling_ratio/min": 0.15668551623821259, "sampling/importance_sampling_ratio/mean": 1.0033835172653198, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49019216001033783, "clip_ratio/low_mean": 0.035245793871581554, "clip_ratio/low_min": 0.035245793871581554, "clip_ratio/high_mean": 0.04776353668421507, "clip_ratio/high_max": 0.04776353668421507, "clip_ratio/region_mean": 0.08300933055579662, "reward_total_mean": 0.5765939354896545, "reward_meter_mean": 0.7750146389007568, "reward_meter_std": 0.2348131388425827, "reward_count_adherence_mean": 0.8958333134651184, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8851284980773926, "reward_repeat_soft_std": 0.09859985113143921, "reward_judge_quality_mean": 0.2462500035762787, "reward_judge_quality_std": 0.09913014620542526, "reward_total_composite_mean": 0.5765939354896545, "reward_total_composite_std": 0.2526305913925171} {"timestamp_utc": "2026-04-13T02:39:39Z", "mode": "train", "global_step": 1732, "epoch": 0.17398292315419386, "loss": 0.0323, "grad_norm": 10.972243309020996, "learning_rate": 4.754545454545455e-06, "num_tokens": 3136069.0, "completions/mean_length": 51.375, "completions/min_length": 48.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.375, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.4373432397842407, "rewards/meter/std": 0.4508380591869354, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9600704908370972, "rewards/repeat_soft/std": 0.024422526359558105, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.606311559677124, "rewards/total_composite/std": 0.17341062426567078, "reward": 0.606311559677124, "reward_std": 0.17341063916683197, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12360525131225586, "sampling/sampling_logp_difference/max": 1.8878672122955322, "sampling/importance_sampling_ratio/min": 0.15139435231685638, "sampling/importance_sampling_ratio/mean": 0.9789642691612244, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48611076176166534, "clip_ratio/low_mean": 0.03986094472929835, "clip_ratio/low_min": 0.03986094472929835, "clip_ratio/high_mean": 0.07013096008449793, "clip_ratio/high_max": 0.07013096008449793, "clip_ratio/region_mean": 0.10999190481379628, "reward_total_mean": 0.606311559677124, "reward_meter_mean": 0.4373432397842407, "reward_meter_std": 0.4508380591869354, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9600704908370972, "reward_repeat_soft_std": 0.024422526359558105, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.606311559677124, "reward_total_composite_std": 0.17341062426567078} {"timestamp_utc": "2026-04-13T02:39:45Z", "mode": "train", "global_step": 1733, "epoch": 0.17408337518834757, "loss": 0.0134, "grad_norm": 13.271747589111328, "learning_rate": 4.751515151515152e-06, "num_tokens": 3137916.0, "completions/mean_length": 52.875, "completions/min_length": 51.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.875, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.7977343797683716, "rewards/meter/std": 0.27399641275405884, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9913161993026733, "rewards/repeat_soft/std": 0.006256903521716595, "rewards/judge_quality/mean": 0.9200000166893005, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8841121196746826, "rewards/total_composite/std": 0.12299244105815887, "reward": 0.8841121196746826, "reward_std": 0.12299242615699768, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10898752510547638, "sampling/sampling_logp_difference/max": 1.441939115524292, "sampling/importance_sampling_ratio/min": 0.23646877706050873, "sampling/importance_sampling_ratio/mean": 1.0151481628417969, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5051665343344212, "clip_ratio/low_mean": 0.038744344376027584, "clip_ratio/low_min": 0.038744344376027584, "clip_ratio/high_mean": 0.07025055447593331, "clip_ratio/high_max": 0.07025055447593331, "clip_ratio/region_mean": 0.1089948988519609, "reward_total_mean": 0.8841121196746826, "reward_meter_mean": 0.7977343797683716, "reward_meter_std": 0.27399641275405884, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9913161993026733, "reward_repeat_soft_std": 0.006256903521716595, "reward_judge_quality_mean": 0.9200000166893005, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8841121196746826, "reward_total_composite_std": 0.12299244105815887} {"timestamp_utc": "2026-04-13T02:39:55Z", "mode": "train", "global_step": 1734, "epoch": 0.17418382722250125, "loss": 0.0396, "grad_norm": 11.825630187988281, "learning_rate": 4.748484848484849e-06, "num_tokens": 3139518.0, "completions/mean_length": 50.25, "completions/min_length": 48.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.25, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.7391326427459717, "rewards/meter/std": 0.40770092606544495, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.951107382774353, "rewards/repeat_soft/std": 0.09018359333276749, "rewards/judge_quality/mean": 0.4300000071525574, "rewards/judge_quality/std": 0.21993505954742432, "rewards/total_composite/mean": 0.7067204713821411, "rewards/total_composite/std": 0.20937348902225494, "reward": 0.7067204713821411, "reward_std": 0.20937347412109375, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11143019795417786, "sampling/sampling_logp_difference/max": 1.5915355682373047, "sampling/importance_sampling_ratio/min": 0.20361271500587463, "sampling/importance_sampling_ratio/mean": 0.9978552460670471, "sampling/importance_sampling_ratio/max": 1.9745415449142456, "entropy": 0.6737916991114616, "clip_ratio/low_mean": 0.03429705183953047, "clip_ratio/low_min": 0.03429705183953047, "clip_ratio/high_mean": 0.10033180704340339, "clip_ratio/high_max": 0.10033180704340339, "clip_ratio/region_mean": 0.13462885888293386, "reward_total_mean": 0.7067204713821411, "reward_meter_mean": 0.7391326427459717, "reward_meter_std": 0.40770092606544495, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.951107382774353, "reward_repeat_soft_std": 0.09018359333276749, "reward_judge_quality_mean": 0.4300000071525574, "reward_judge_quality_std": 0.21993505954742432, "reward_total_composite_mean": 0.7067204713821411, "reward_total_composite_std": 0.20937348902225494} {"timestamp_utc": "2026-04-13T02:40:01Z", "mode": "train", "global_step": 1735, "epoch": 0.17428427925665493, "loss": -0.001, "grad_norm": 11.038545608520508, "learning_rate": 4.745454545454546e-06, "num_tokens": 3141084.0, "completions/mean_length": 45.75, "completions/min_length": 39.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.75, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.9884481430053711, "rewards/meter/std": 0.007377295289188623, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9253838062286377, "rewards/repeat_soft/std": 0.044106271117925644, "rewards/judge_quality/mean": 0.5024999976158142, "rewards/judge_quality/std": 0.1440981924533844, "rewards/total_composite/mean": 0.8380900025367737, "rewards/total_composite/std": 0.04263549670577049, "reward": 0.8380900025367737, "reward_std": 0.042635489255189896, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09054736793041229, "sampling/sampling_logp_difference/max": 1.2472072839736938, "sampling/importance_sampling_ratio/min": 0.2873060703277588, "sampling/importance_sampling_ratio/mean": 1.022813320159912, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48739833384752274, "clip_ratio/low_mean": 0.08277224376797676, "clip_ratio/low_min": 0.08277224376797676, "clip_ratio/high_mean": 0.023639456368982792, "clip_ratio/high_max": 0.023639456368982792, "clip_ratio/region_mean": 0.10641170013695955, "reward_total_mean": 0.8380900025367737, "reward_meter_mean": 0.9884481430053711, "reward_meter_std": 0.007377295289188623, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9253838062286377, "reward_repeat_soft_std": 0.044106271117925644, "reward_judge_quality_mean": 0.5024999976158142, "reward_judge_quality_std": 0.1440981924533844, "reward_total_composite_mean": 0.8380900025367737, "reward_total_composite_std": 0.04263549670577049} {"timestamp_utc": "2026-04-13T02:40:08Z", "mode": "train", "global_step": 1736, "epoch": 0.17438473129080864, "loss": -0.0124, "grad_norm": 11.384105682373047, "learning_rate": 4.7424242424242426e-06, "num_tokens": 3143312.0, "completions/mean_length": 92.5, "completions/min_length": 83.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.5, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.8456052541732788, "rewards/meter/std": 0.20563486218452454, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9024255871772766, "rewards/repeat_soft/std": 0.08478915691375732, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7420774698257446, "rewards/total_composite/std": 0.0973232090473175, "reward": 0.7420774698257446, "reward_std": 0.0973232090473175, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09689605236053467, "sampling/sampling_logp_difference/max": 1.6395187377929688, "sampling/importance_sampling_ratio/min": 0.19407342374324799, "sampling/importance_sampling_ratio/mean": 1.0083891153335571, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5091359466314316, "clip_ratio/low_mean": 0.02696360182017088, "clip_ratio/low_min": 0.02696360182017088, "clip_ratio/high_mean": 0.07106643682345748, "clip_ratio/high_max": 0.07106643682345748, "clip_ratio/region_mean": 0.09803003864362836, "reward_total_mean": 0.7420774698257446, "reward_meter_mean": 0.8456052541732788, "reward_meter_std": 0.20563486218452454, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9024255871772766, "reward_repeat_soft_std": 0.08478915691375732, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7420774698257446, "reward_total_composite_std": 0.0973232090473175} {"timestamp_utc": "2026-04-13T02:40:15Z", "mode": "train", "global_step": 1737, "epoch": 0.17448518332496232, "loss": 0.0001, "grad_norm": 9.06169605255127, "learning_rate": 4.73939393939394e-06, "num_tokens": 3144982.0, "completions/mean_length": 54.75, "completions/min_length": 49.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.75, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9088954925537109, "rewards/meter/std": 0.16764843463897705, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.965968906879425, "rewards/repeat_soft/std": 0.029489241540431976, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7815998792648315, "rewards/total_composite/std": 0.07375601679086685, "reward": 0.7815998792648315, "reward_std": 0.07375600188970566, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13502587378025055, "sampling/sampling_logp_difference/max": 2.8635504245758057, "sampling/importance_sampling_ratio/min": 0.05706579610705376, "sampling/importance_sampling_ratio/mean": 0.997163712978363, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7045829705893993, "clip_ratio/low_mean": 0.033566057682037354, "clip_ratio/low_min": 0.033566057682037354, "clip_ratio/high_mean": 0.1011765138246119, "clip_ratio/high_max": 0.1011765138246119, "clip_ratio/region_mean": 0.13474257150664926, "reward_total_mean": 0.7815998792648315, "reward_meter_mean": 0.9088954925537109, "reward_meter_std": 0.16764843463897705, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.965968906879425, "reward_repeat_soft_std": 0.029489241540431976, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7815998792648315, "reward_total_composite_std": 0.07375601679086685} {"timestamp_utc": "2026-04-13T02:40:22Z", "mode": "train", "global_step": 1738, "epoch": 0.17458563535911603, "loss": 0.0237, "grad_norm": 11.064888000488281, "learning_rate": 4.736363636363637e-06, "num_tokens": 3147452.0, "completions/mean_length": 97.75, "completions/min_length": 94.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.75, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.9219661951065063, "rewards/meter/std": 0.14073768258094788, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8742314577102661, "rewards/repeat_soft/std": 0.03868637979030609, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7783079743385315, "rewards/total_composite/std": 0.06345173716545105, "reward": 0.7783079743385315, "reward_std": 0.06345174461603165, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11542820930480957, "sampling/sampling_logp_difference/max": 3.149907112121582, "sampling/importance_sampling_ratio/min": 0.04285610839724541, "sampling/importance_sampling_ratio/mean": 1.0000299215316772, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3543597385287285, "clip_ratio/low_mean": 0.03223726712167263, "clip_ratio/low_min": 0.03223726712167263, "clip_ratio/high_mean": 0.07394188828766346, "clip_ratio/high_max": 0.07394188828766346, "clip_ratio/region_mean": 0.10617915540933609, "reward_total_mean": 0.7783079743385315, "reward_meter_mean": 0.9219661951065063, "reward_meter_std": 0.14073768258094788, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8742314577102661, "reward_repeat_soft_std": 0.03868637979030609, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7783079743385315, "reward_total_composite_std": 0.06345173716545105} {"timestamp_utc": "2026-04-13T02:40:29Z", "mode": "train", "global_step": 1739, "epoch": 0.1746860873932697, "loss": -0.0063, "grad_norm": 16.824953079223633, "learning_rate": 4.7333333333333335e-06, "num_tokens": 3149043.0, "completions/mean_length": 33.875, "completions/min_length": 28.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.875, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.39587581157684326, "rewards/meter/std": 0.32419610023498535, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8944485187530518, "rewards/repeat_soft/std": 0.12878550589084625, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404937028885, "rewards/total_composite/mean": 0.5844639539718628, "rewards/total_composite/std": 0.16455377638339996, "reward": 0.5844639539718628, "reward_std": 0.16455377638339996, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11342363059520721, "sampling/sampling_logp_difference/max": 1.921218991279602, "sampling/importance_sampling_ratio/min": 0.14642836153507233, "sampling/importance_sampling_ratio/mean": 1.0024263858795166, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48570191860198975, "clip_ratio/low_mean": 0.042029738426208496, "clip_ratio/low_min": 0.042029738426208496, "clip_ratio/high_mean": 0.049554409459233284, "clip_ratio/high_max": 0.049554409459233284, "clip_ratio/region_mean": 0.09158414788544178, "reward_total_mean": 0.5844639539718628, "reward_meter_mean": 0.39587581157684326, "reward_meter_std": 0.32419610023498535, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8944485187530518, "reward_repeat_soft_std": 0.12878550589084625, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404937028885, "reward_total_composite_mean": 0.5844639539718628, "reward_total_composite_std": 0.16455377638339996} {"timestamp_utc": "2026-04-13T02:40:36Z", "mode": "train", "global_step": 1740, "epoch": 0.1747865394274234, "loss": 0.0144, "grad_norm": 6.842529296875, "learning_rate": 4.730303030303031e-06, "num_tokens": 3151320.0, "completions/mean_length": 97.625, "completions/min_length": 87.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.625, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.9893338680267334, "rewards/meter/std": 0.003009534440934658, "rewards/count_adherence/mean": 0.8500000238418579, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6382988691329956, "rewards/repeat_soft/std": 0.12600597739219666, "rewards/judge_quality/mean": 0.41874998807907104, "rewards/judge_quality/std": 0.21931305527687073, "rewards/total_composite/mean": 0.7621551156044006, "rewards/total_composite/std": 0.06295330077409744, "reward": 0.7621551156044006, "reward_std": 0.06295328587293625, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09984464943408966, "sampling/sampling_logp_difference/max": 6.246114730834961, "sampling/importance_sampling_ratio/min": 0.0019379690056666732, "sampling/importance_sampling_ratio/mean": 1.0000158548355103, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4119758643209934, "clip_ratio/low_mean": 0.03971618693321943, "clip_ratio/low_min": 0.03971618693321943, "clip_ratio/high_mean": 0.032045419327914715, "clip_ratio/high_max": 0.032045419327914715, "clip_ratio/region_mean": 0.07176160626113415, "reward_total_mean": 0.7621551156044006, "reward_meter_mean": 0.9893338680267334, "reward_meter_std": 0.003009534440934658, "reward_count_adherence_mean": 0.8500000238418579, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6382988691329956, "reward_repeat_soft_std": 0.12600597739219666, "reward_judge_quality_mean": 0.41874998807907104, "reward_judge_quality_std": 0.21931305527687073, "reward_total_composite_mean": 0.7621551156044006, "reward_total_composite_std": 0.06295330077409744} {"timestamp_utc": "2026-04-13T02:40:48Z", "mode": "train", "global_step": 1741, "epoch": 0.1748869914615771, "loss": -0.1103, "grad_norm": 3.3650686740875244, "learning_rate": 4.727272727272728e-06, "num_tokens": 3152953.0, "completions/mean_length": 105.125, "completions/min_length": 43.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 47.000003814697266, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.49645018577575684, "rewards/meter/std": 0.3927074372768402, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9916478991508484, "rewards/repeat_soft/std": 0.008574100211262703, "rewards/judge_quality/mean": 0.4437500238418579, "rewards/judge_quality/std": 0.23427321016788483, "rewards/total_composite/mean": 0.5725674033164978, "rewards/total_composite/std": 0.2692025303840637, "reward": 0.5725674033164978, "reward_std": 0.2692025601863861, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1196431815624237, "sampling/sampling_logp_difference/max": 1.679535984992981, "sampling/importance_sampling_ratio/min": 0.186460480093956, "sampling/importance_sampling_ratio/mean": 1.0186148881912231, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4799778312444687, "clip_ratio/low_mean": 0.013586956076323986, "clip_ratio/low_min": 0.013586956076323986, "clip_ratio/high_mean": 0.09293372370302677, "clip_ratio/high_max": 0.09293372370302677, "clip_ratio/region_mean": 0.10652067977935076, "reward_total_mean": 0.5725674033164978, "reward_meter_mean": 0.49645018577575684, "reward_meter_std": 0.3927074372768402, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9916478991508484, "reward_repeat_soft_std": 0.008574100211262703, "reward_judge_quality_mean": 0.4437500238418579, "reward_judge_quality_std": 0.23427321016788483, "reward_total_composite_mean": 0.5725674033164978, "reward_total_composite_std": 0.2692025303840637} {"timestamp_utc": "2026-04-13T02:40:56Z", "mode": "train", "global_step": 1742, "epoch": 0.17498744349573078, "loss": 0.0689, "grad_norm": 5.5244011878967285, "learning_rate": 4.724242424242424e-06, "num_tokens": 3155621.0, "completions/mean_length": 145.5, "completions/min_length": 124.0, "completions/max_length": 175.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 145.5, "completions/min_terminated_length": 124.0, "completions/max_terminated_length": 175.0, "rewards/meter/mean": 0.8924340605735779, "rewards/meter/std": 0.09305308759212494, "rewards/count_adherence/mean": 0.9166666269302368, "rewards/count_adherence/std": 0.0890870913863182, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6892834901809692, "rewards/repeat_soft/std": 0.09273386001586914, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.7715237140655518, "rewards/total_composite/std": 0.07315706461668015, "reward": 0.7715237140655518, "reward_std": 0.07315704226493835, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07729364931583405, "sampling/sampling_logp_difference/max": 2.3403217792510986, "sampling/importance_sampling_ratio/min": 0.09629664570093155, "sampling/importance_sampling_ratio/mean": 0.999067485332489, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.376950666308403, "clip_ratio/low_mean": 0.030505204806104302, "clip_ratio/low_min": 0.030505204806104302, "clip_ratio/high_mean": 0.024438622407615185, "clip_ratio/high_max": 0.024438622407615185, "clip_ratio/region_mean": 0.05494382721371949, "reward_total_mean": 0.7715237140655518, "reward_meter_mean": 0.8924340605735779, "reward_meter_std": 0.09305308759212494, "reward_count_adherence_mean": 0.9166666269302368, "reward_count_adherence_std": 0.0890870913863182, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6892834901809692, "reward_repeat_soft_std": 0.09273386001586914, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.7715237140655518, "reward_total_composite_std": 0.07315706461668015} {"timestamp_utc": "2026-04-13T02:41:08Z", "mode": "train", "global_step": 1743, "epoch": 0.1750878955298845, "loss": -0.1341, "grad_norm": 1.788570761680603, "learning_rate": 4.721212121212122e-06, "num_tokens": 3157267.0, "completions/mean_length": 106.75, "completions/min_length": 45.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 48.85714340209961, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9881476759910583, "rewards/meter/std": 0.008460946381092072, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9169219732284546, "rewards/repeat_soft/std": 0.058509912341833115, "rewards/judge_quality/mean": 0.3187499940395355, "rewards/judge_quality/std": 0.14961259067058563, "rewards/total_composite/mean": 0.6930865049362183, "rewards/total_composite/std": 0.28170037269592285, "reward": 0.6930865049362183, "reward_std": 0.28170037269592285, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12836304306983948, "sampling/sampling_logp_difference/max": 1.563621997833252, "sampling/importance_sampling_ratio/min": 0.20937632024288177, "sampling/importance_sampling_ratio/mean": 1.001489281654358, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.557272233068943, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.1389586115255952, "clip_ratio/high_max": 0.1389586115255952, "clip_ratio/region_mean": 0.1389586115255952, "reward_total_mean": 0.6930865049362183, "reward_meter_mean": 0.9881476759910583, "reward_meter_std": 0.008460946381092072, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9169219732284546, "reward_repeat_soft_std": 0.058509912341833115, "reward_judge_quality_mean": 0.3187499940395355, "reward_judge_quality_std": 0.14961259067058563, "reward_total_composite_mean": 0.6930865049362183, "reward_total_composite_std": 0.28170037269592285} {"timestamp_utc": "2026-04-13T02:41:19Z", "mode": "train", "global_step": 1744, "epoch": 0.17518834756403817, "loss": -0.0795, "grad_norm": 3.0901248455047607, "learning_rate": 4.718181818181818e-06, "num_tokens": 3158621.0, "completions/mean_length": 89.25, "completions/min_length": 23.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 28.85714340209961, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.5119338035583496, "rewards/meter/std": 0.42880284786224365, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8812848329544067, "rewards/repeat_soft/std": 0.14372454583644867, "rewards/judge_quality/mean": 0.35874998569488525, "rewards/judge_quality/std": 0.16225530207157135, "rewards/total_composite/mean": 0.5429986715316772, "rewards/total_composite/std": 0.2717478275299072, "reward": 0.5429986715316772, "reward_std": 0.2717478275299072, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14132088422775269, "sampling/sampling_logp_difference/max": 1.4746966361999512, "sampling/importance_sampling_ratio/min": 0.2288481593132019, "sampling/importance_sampling_ratio/mean": 1.0233105421066284, "sampling/importance_sampling_ratio/max": 1.9356290102005005, "entropy": 0.8362426832318306, "clip_ratio/low_mean": 0.0612903218716383, "clip_ratio/low_min": 0.0612903218716383, "clip_ratio/high_mean": 0.07073534652590752, "clip_ratio/high_max": 0.07073534652590752, "clip_ratio/region_mean": 0.13202566839754581, "reward_total_mean": 0.5429986715316772, "reward_meter_mean": 0.5119338035583496, "reward_meter_std": 0.42880284786224365, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8812848329544067, "reward_repeat_soft_std": 0.14372454583644867, "reward_judge_quality_mean": 0.35874998569488525, "reward_judge_quality_std": 0.16225530207157135, "reward_total_composite_mean": 0.5429986715316772, "reward_total_composite_std": 0.2717478275299072} {"timestamp_utc": "2026-04-13T02:41:25Z", "mode": "train", "global_step": 1745, "epoch": 0.17528879959819185, "loss": 0.0651, "grad_norm": 18.208555221557617, "learning_rate": 4.715151515151515e-06, "num_tokens": 3160086.0, "completions/mean_length": 28.125, "completions/min_length": 25.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.125, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.7700051069259644, "rewards/meter/std": 0.36170926690101624, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9610613584518433, "rewards/repeat_soft/std": 0.004068940877914429, "rewards/judge_quality/mean": 0.5087499618530273, "rewards/judge_quality/std": 0.16617010533809662, "rewards/total_composite/mean": 0.7452334761619568, "rewards/total_composite/std": 0.12254638969898224, "reward": 0.7452334761619568, "reward_std": 0.12254638969898224, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12792816758155823, "sampling/sampling_logp_difference/max": 1.1741809844970703, "sampling/importance_sampling_ratio/min": 0.33290016651153564, "sampling/importance_sampling_ratio/mean": 1.0117682218551636, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7334705591201782, "clip_ratio/low_mean": 0.028544372878968716, "clip_ratio/low_min": 0.028544372878968716, "clip_ratio/high_mean": 0.07923776376992464, "clip_ratio/high_max": 0.07923776376992464, "clip_ratio/region_mean": 0.10778213664889336, "reward_total_mean": 0.7452334761619568, "reward_meter_mean": 0.7700051069259644, "reward_meter_std": 0.36170926690101624, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9610613584518433, "reward_repeat_soft_std": 0.004068940877914429, "reward_judge_quality_mean": 0.5087499618530273, "reward_judge_quality_std": 0.16617010533809662, "reward_total_composite_mean": 0.7452334761619568, "reward_total_composite_std": 0.12254638969898224} {"timestamp_utc": "2026-04-13T02:41:33Z", "mode": "train", "global_step": 1746, "epoch": 0.17538925163234556, "loss": -0.0113, "grad_norm": 5.912535667419434, "learning_rate": 4.7121212121212126e-06, "num_tokens": 3162974.0, "completions/mean_length": 147.0, "completions/min_length": 134.0, "completions/max_length": 164.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 147.0, "completions/min_terminated_length": 134.0, "completions/max_terminated_length": 164.0, "rewards/meter/mean": 0.991820216178894, "rewards/meter/std": 0.006107361521571875, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8792431950569153, "rewards/repeat_soft/std": 0.05381795018911362, "rewards/judge_quality/mean": 0.26749998331069946, "rewards/judge_quality/std": 0.10375107079744339, "rewards/total_composite/mean": 0.7607433795928955, "rewards/total_composite/std": 0.03511416167020798, "reward": 0.7607433795928955, "reward_std": 0.035114143043756485, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0963420644402504, "sampling/sampling_logp_difference/max": 1.6771202087402344, "sampling/importance_sampling_ratio/min": 0.1869114637374878, "sampling/importance_sampling_ratio/mean": 1.0047762393951416, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5092811398208141, "clip_ratio/low_mean": 0.045702168717980385, "clip_ratio/low_min": 0.045702168717980385, "clip_ratio/high_mean": 0.05781740043312311, "clip_ratio/high_max": 0.05781740043312311, "clip_ratio/region_mean": 0.1035195691511035, "reward_total_mean": 0.7607433795928955, "reward_meter_mean": 0.991820216178894, "reward_meter_std": 0.006107361521571875, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8792431950569153, "reward_repeat_soft_std": 0.05381795018911362, "reward_judge_quality_mean": 0.26749998331069946, "reward_judge_quality_std": 0.10375107079744339, "reward_total_composite_mean": 0.7607433795928955, "reward_total_composite_std": 0.03511416167020798} {"timestamp_utc": "2026-04-13T02:41:41Z", "mode": "train", "global_step": 1747, "epoch": 0.17548970366649924, "loss": 0.0051, "grad_norm": 8.353354454040527, "learning_rate": 4.709090909090909e-06, "num_tokens": 3165360.0, "completions/mean_length": 107.25, "completions/min_length": 96.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.25, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.6929640769958496, "rewards/meter/std": 0.33879542350769043, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9530451893806458, "rewards/repeat_soft/std": 0.04414280131459236, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.6703883409500122, "rewards/total_composite/std": 0.1427982598543167, "reward": 0.6703883409500122, "reward_std": 0.1427982747554779, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11488506197929382, "sampling/sampling_logp_difference/max": 1.5605812072753906, "sampling/importance_sampling_ratio/min": 0.2100139707326889, "sampling/importance_sampling_ratio/mean": 1.0070816278457642, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5935643501579762, "clip_ratio/low_mean": 0.04838003218173981, "clip_ratio/low_min": 0.04838003218173981, "clip_ratio/high_mean": 0.06836645863950253, "clip_ratio/high_max": 0.06836645863950253, "clip_ratio/region_mean": 0.11674649082124233, "reward_total_mean": 0.6703883409500122, "reward_meter_mean": 0.6929640769958496, "reward_meter_std": 0.33879542350769043, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9530451893806458, "reward_repeat_soft_std": 0.04414280131459236, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.6703883409500122, "reward_total_composite_std": 0.1427982598543167} {"timestamp_utc": "2026-04-13T02:41:47Z", "mode": "train", "global_step": 1748, "epoch": 0.17559015570065295, "loss": 0.0698, "grad_norm": 12.070812225341797, "learning_rate": 4.706060606060606e-06, "num_tokens": 3167132.0, "completions/mean_length": 54.5, "completions/min_length": 46.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.5, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.8995468020439148, "rewards/meter/std": 0.15441565215587616, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9722813367843628, "rewards/repeat_soft/std": 0.026508556678891182, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.10260014235973358, "rewards/total_composite/mean": 0.7926491498947144, "rewards/total_composite/std": 0.07552090287208557, "reward": 0.7926491498947144, "reward_std": 0.07552091032266617, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12728503346443176, "sampling/sampling_logp_difference/max": 2.0329132080078125, "sampling/importance_sampling_ratio/min": 0.13095347583293915, "sampling/importance_sampling_ratio/mean": 1.0070229768753052, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.745768629014492, "clip_ratio/low_mean": 0.023373016621917486, "clip_ratio/low_min": 0.023373016621917486, "clip_ratio/high_mean": 0.09414185862988234, "clip_ratio/high_max": 0.09414185862988234, "clip_ratio/region_mean": 0.11751487525179982, "reward_total_mean": 0.7926491498947144, "reward_meter_mean": 0.8995468020439148, "reward_meter_std": 0.15441565215587616, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9722813367843628, "reward_repeat_soft_std": 0.026508556678891182, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.10260014235973358, "reward_total_composite_mean": 0.7926491498947144, "reward_total_composite_std": 0.07552090287208557} {"timestamp_utc": "2026-04-13T02:41:54Z", "mode": "train", "global_step": 1749, "epoch": 0.17569060773480663, "loss": 0.0355, "grad_norm": 15.26647663116455, "learning_rate": 4.7030303030303035e-06, "num_tokens": 3168516.0, "completions/mean_length": 26.0, "completions/min_length": 19.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.0, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9812400341033936, "rewards/meter/std": 0.01239354070276022, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9558870196342468, "rewards/repeat_soft/std": 0.018704308196902275, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.08166787773370743, "rewards/total_composite/mean": 0.8015217185020447, "rewards/total_composite/std": 0.030455294996500015, "reward": 0.8015217185020447, "reward_std": 0.030455317348241806, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11526919156312943, "sampling/sampling_logp_difference/max": 0.9130349159240723, "sampling/importance_sampling_ratio/min": 0.4013044834136963, "sampling/importance_sampling_ratio/mean": 1.012918472290039, "sampling/importance_sampling_ratio/max": 1.9726179838180542, "entropy": 0.7768649905920029, "clip_ratio/low_mean": 0.03209677338600159, "clip_ratio/low_min": 0.03209677338600159, "clip_ratio/high_mean": 0.07859967602416873, "clip_ratio/high_max": 0.07859967602416873, "clip_ratio/region_mean": 0.11069644941017032, "reward_total_mean": 0.8015217185020447, "reward_meter_mean": 0.9812400341033936, "reward_meter_std": 0.01239354070276022, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9558870196342468, "reward_repeat_soft_std": 0.018704308196902275, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.08166787773370743, "reward_total_composite_mean": 0.8015217185020447, "reward_total_composite_std": 0.030455294996500015} {"timestamp_utc": "2026-04-13T02:42:00Z", "mode": "train", "global_step": 1750, "epoch": 0.1757910597689603, "loss": 0.018, "grad_norm": 10.541484832763672, "learning_rate": 4.7e-06, "num_tokens": 3170740.0, "completions/mean_length": 97.0, "completions/min_length": 92.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.0, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.6922869682312012, "rewards/meter/std": 0.3359096646308899, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9201050996780396, "rewards/repeat_soft/std": 0.052849020808935165, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.6667896509170532, "rewards/total_composite/std": 0.15466226637363434, "reward": 0.6667896509170532, "reward_std": 0.15466228127479553, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0964021161198616, "sampling/sampling_logp_difference/max": 1.5896263122558594, "sampling/importance_sampling_ratio/min": 0.20400182902812958, "sampling/importance_sampling_ratio/mean": 1.0063180923461914, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5722656697034836, "clip_ratio/low_mean": 0.030074188951402903, "clip_ratio/low_min": 0.030074188951402903, "clip_ratio/high_mean": 0.06997561501339078, "clip_ratio/high_max": 0.06997561501339078, "clip_ratio/region_mean": 0.10004980396479368, "reward_total_mean": 0.6667896509170532, "reward_meter_mean": 0.6922869682312012, "reward_meter_std": 0.3359096646308899, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9201050996780396, "reward_repeat_soft_std": 0.052849020808935165, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.6667896509170532, "reward_total_composite_std": 0.15466226637363434} {"timestamp_utc": "2026-04-13T02:42:49Z", "mode": "eval", "global_step": 1750, "epoch": 0.1757910597689603, "eval_loss": NaN, "eval_runtime": 48.0784, "eval_samples_per_second": 1.664, "eval_steps_per_second": 0.208, "eval_num_tokens": 3170740.0, "eval_completions/mean_length": 86.4125, "eval_completions/min_length": 39.4, "eval_completions/max_length": 172.9, "eval_completions/clipped_ratio": 0.0125, "eval_completions/mean_terminated_length": 80.9125, "eval_completions/min_terminated_length": 39.4, "eval_completions/max_terminated_length": 135.1, "eval_rewards/meter/mean": 0.7211499065160751, "eval_rewards/meter/std": 0.35507472008466723, "eval_rewards/count_adherence/mean": 0.9885416626930237, "eval_rewards/count_adherence/std": 0.03240906037390232, "eval_rewards/hard_gate/mean": 0.9875, "eval_rewards/hard_gate/std": 0.03535533845424652, "eval_rewards/repeat_soft/mean": 0.9048998475074768, "eval_rewards/repeat_soft/std": 0.08222445547580719, "eval_rewards/judge_quality/mean": 0.4106249988079071, "eval_rewards/judge_quality/std": 0.12404979774728417, "eval_rewards/total_composite/mean": 0.6846165299415589, "eval_rewards/total_composite/std": 0.1676706351339817, "eval_reward": 0.6846165299415589, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.048794782161712645, "eval_sampling/sampling_logp_difference/max": 0.8487237453460693, "eval_sampling/importance_sampling_ratio/min": 0.43413779735565183, "eval_sampling/importance_sampling_ratio/mean": 1.0104714512825013, "eval_sampling/importance_sampling_ratio/max": 1.37550288438797, "eval_entropy": 0.5264610946178436, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6846165299415589, "eval_reward_meter_mean": 0.7211499065160751, "eval_reward_meter_std": 0.35507472008466723, "eval_reward_count_adherence_mean": 0.9885416626930237, "eval_reward_count_adherence_std": 0.03240906037390232, "eval_reward_hard_gate_mean": 0.9875, "eval_reward_hard_gate_std": 0.03535533845424652, "eval_reward_repeat_soft_mean": 0.9048998475074768, "eval_reward_repeat_soft_std": 0.08222445547580719, "eval_reward_judge_quality_mean": 0.4106249988079071, "eval_reward_judge_quality_std": 0.12404979774728417, "eval_reward_total_composite_mean": 0.6846165299415589, "eval_reward_total_composite_std": 0.1676706351339817} {"timestamp_utc": "2026-04-13T02:42:58Z", "mode": "train", "global_step": 1751, "epoch": 0.17589151180311402, "loss": 0.088, "grad_norm": 8.080245971679688, "learning_rate": 4.696969696969698e-06, "num_tokens": 3173128.0, "completions/mean_length": 122.5, "completions/min_length": 109.0, "completions/max_length": 141.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.5, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 141.0, "rewards/meter/mean": 0.965452253818512, "rewards/meter/std": 0.06345534324645996, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7365334033966064, "rewards/repeat_soft/std": 0.13738228380680084, "rewards/judge_quality/mean": 0.29249998927116394, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7421068549156189, "rewards/total_composite/std": 0.05099770426750183, "reward": 0.7421068549156189, "reward_std": 0.05099770054221153, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09435416758060455, "sampling/sampling_logp_difference/max": 1.8767507076263428, "sampling/importance_sampling_ratio/min": 0.15308672189712524, "sampling/importance_sampling_ratio/mean": 1.0167633295059204, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5218653120100498, "clip_ratio/low_mean": 0.03362002782523632, "clip_ratio/low_min": 0.03362002782523632, "clip_ratio/high_mean": 0.04119821544736624, "clip_ratio/high_max": 0.04119821544736624, "clip_ratio/region_mean": 0.07481824327260256, "reward_total_mean": 0.7421068549156189, "reward_meter_mean": 0.965452253818512, "reward_meter_std": 0.06345534324645996, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7365334033966064, "reward_repeat_soft_std": 0.13738228380680084, "reward_judge_quality_mean": 0.29249998927116394, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7421068549156189, "reward_total_composite_std": 0.05099770426750183} {"timestamp_utc": "2026-04-13T02:43:10Z", "mode": "train", "global_step": 1752, "epoch": 0.1759919638372677, "loss": -0.1025, "grad_norm": 3.0249228477478027, "learning_rate": 4.693939393939394e-06, "num_tokens": 3174577.0, "completions/mean_length": 98.125, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 39.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.6688722372055054, "rewards/meter/std": 0.34564486145973206, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9578362107276917, "rewards/repeat_soft/std": 0.08781595528125763, "rewards/judge_quality/mean": 0.6100000143051147, "rewards/judge_quality/std": 0.3543202877044678, "rewards/total_composite/mean": 0.6966511011123657, "rewards/total_composite/std": 0.311928391456604, "reward": 0.6966511011123657, "reward_std": 0.311928391456604, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13689182698726654, "sampling/sampling_logp_difference/max": 1.3201324939727783, "sampling/importance_sampling_ratio/min": 0.26709991693496704, "sampling/importance_sampling_ratio/mean": 0.9646511673927307, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4659160375595093, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/high_mean": 0.11452938755974174, "clip_ratio/high_max": 0.11452938755974174, "clip_ratio/region_mean": 0.11790776601992548, "reward_total_mean": 0.6966511011123657, "reward_meter_mean": 0.6688722372055054, "reward_meter_std": 0.34564486145973206, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9578362107276917, "reward_repeat_soft_std": 0.08781595528125763, "reward_judge_quality_mean": 0.6100000143051147, "reward_judge_quality_std": 0.3543202877044678, "reward_total_composite_mean": 0.6966511011123657, "reward_total_composite_std": 0.311928391456604} {"timestamp_utc": "2026-04-13T02:43:17Z", "mode": "train", "global_step": 1753, "epoch": 0.1760924158714214, "loss": 0.0675, "grad_norm": 19.139305114746094, "learning_rate": 4.690909090909092e-06, "num_tokens": 3175956.0, "completions/mean_length": 24.375, "completions/min_length": 20.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.375, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.957506000995636, "rewards/meter/std": 0.05830962583422661, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9594928026199341, "rewards/repeat_soft/std": 0.008505655452609062, "rewards/judge_quality/mean": 0.6187499761581421, "rewards/judge_quality/std": 0.24976776540279388, "rewards/total_composite/mean": 0.8624520301818848, "rewards/total_composite/std": 0.07940862327814102, "reward": 0.8624520301818848, "reward_std": 0.07940862327814102, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1304331123828888, "sampling/sampling_logp_difference/max": 1.9883394241333008, "sampling/importance_sampling_ratio/min": 0.13692261278629303, "sampling/importance_sampling_ratio/mean": 1.0024940967559814, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7049298360943794, "clip_ratio/low_mean": 0.04611557023599744, "clip_ratio/low_min": 0.04611557023599744, "clip_ratio/high_mean": 0.042412207927554846, "clip_ratio/high_max": 0.042412207927554846, "clip_ratio/region_mean": 0.08852777816355228, "reward_total_mean": 0.8624520301818848, "reward_meter_mean": 0.957506000995636, "reward_meter_std": 0.05830962583422661, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9594928026199341, "reward_repeat_soft_std": 0.008505655452609062, "reward_judge_quality_mean": 0.6187499761581421, "reward_judge_quality_std": 0.24976776540279388, "reward_total_composite_mean": 0.8624520301818848, "reward_total_composite_std": 0.07940862327814102} {"timestamp_utc": "2026-04-13T02:43:28Z", "mode": "train", "global_step": 1754, "epoch": 0.1761928679055751, "loss": -0.1377, "grad_norm": 2.7674453258514404, "learning_rate": 4.687878787878788e-06, "num_tokens": 3177694.0, "completions/mean_length": 127.25, "completions/min_length": 65.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 72.28572082519531, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.42375701665878296, "rewards/meter/std": 0.33155882358551025, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9716469049453735, "rewards/repeat_soft/std": 0.019774312153458595, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.13617216050624847, "rewards/total_composite/mean": 0.5194966793060303, "rewards/total_composite/std": 0.245919868350029, "reward": 0.5194966793060303, "reward_std": 0.2459198534488678, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12714065611362457, "sampling/sampling_logp_difference/max": 1.5289708375930786, "sampling/importance_sampling_ratio/min": 0.2167586386203766, "sampling/importance_sampling_ratio/mean": 1.0094189643859863, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5634930059313774, "clip_ratio/low_mean": 0.05185030214488506, "clip_ratio/low_min": 0.05185030214488506, "clip_ratio/high_mean": 0.059566321317106485, "clip_ratio/high_max": 0.059566321317106485, "clip_ratio/region_mean": 0.11141662346199155, "reward_total_mean": 0.5194966793060303, "reward_meter_mean": 0.42375701665878296, "reward_meter_std": 0.33155882358551025, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9716469049453735, "reward_repeat_soft_std": 0.019774312153458595, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.13617216050624847, "reward_total_composite_mean": 0.5194966793060303, "reward_total_composite_std": 0.245919868350029} {"timestamp_utc": "2026-04-13T02:43:39Z", "mode": "train", "global_step": 1755, "epoch": 0.17629331993972877, "loss": -0.0093, "grad_norm": 6.319343090057373, "learning_rate": 4.684848484848485e-06, "num_tokens": 3180146.0, "completions/mean_length": 115.5, "completions/min_length": 104.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.5, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.8702743649482727, "rewards/meter/std": 0.18253663182258606, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9244818687438965, "rewards/repeat_soft/std": 0.05078219249844551, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.747321605682373, "rewards/total_composite/std": 0.07645302265882492, "reward": 0.747321605682373, "reward_std": 0.07645303010940552, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11028116196393967, "sampling/sampling_logp_difference/max": 1.7101459503173828, "sampling/importance_sampling_ratio/min": 0.180839404463768, "sampling/importance_sampling_ratio/mean": 1.0127065181732178, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7665686793625355, "clip_ratio/low_mean": 0.030842390842735767, "clip_ratio/low_min": 0.030842390842735767, "clip_ratio/high_mean": 0.07950984826311469, "clip_ratio/high_max": 0.07950984826311469, "clip_ratio/region_mean": 0.11035223910585046, "reward_total_mean": 0.747321605682373, "reward_meter_mean": 0.8702743649482727, "reward_meter_std": 0.18253663182258606, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9244818687438965, "reward_repeat_soft_std": 0.05078219249844551, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.747321605682373, "reward_total_composite_std": 0.07645302265882492} {"timestamp_utc": "2026-04-13T02:43:47Z", "mode": "train", "global_step": 1756, "epoch": 0.17639377197388248, "loss": -0.0161, "grad_norm": 9.04842758178711, "learning_rate": 4.681818181818183e-06, "num_tokens": 3182244.0, "completions/mean_length": 95.25, "completions/min_length": 89.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.25, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.5154112577438354, "rewards/meter/std": 0.383279412984848, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9278315305709839, "rewards/repeat_soft/std": 0.061372701078653336, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.5103014707565308, "rewards/total_composite/std": 0.25096043944358826, "reward": 0.5103014707565308, "reward_std": 0.25096040964126587, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09678500890731812, "sampling/sampling_logp_difference/max": 1.349776029586792, "sampling/importance_sampling_ratio/min": 0.25929832458496094, "sampling/importance_sampling_ratio/mean": 1.0036307573318481, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49755600467324257, "clip_ratio/low_mean": 0.038422953337430954, "clip_ratio/low_min": 0.038422953337430954, "clip_ratio/high_mean": 0.07283303141593933, "clip_ratio/high_max": 0.07283303141593933, "clip_ratio/region_mean": 0.11125598475337029, "reward_total_mean": 0.5103014707565308, "reward_meter_mean": 0.5154112577438354, "reward_meter_std": 0.383279412984848, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9278315305709839, "reward_repeat_soft_std": 0.061372701078653336, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.5103014707565308, "reward_total_composite_std": 0.25096043944358826} {"timestamp_utc": "2026-04-13T02:43:58Z", "mode": "train", "global_step": 1757, "epoch": 0.17649422400803616, "loss": 0.0645, "grad_norm": 28.255342483520508, "learning_rate": 4.678787878787879e-06, "num_tokens": 3183651.0, "completions/mean_length": 24.875, "completions/min_length": 20.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.875, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9814830422401428, "rewards/meter/std": 0.02102908305823803, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9503509998321533, "rewards/repeat_soft/std": 0.03320911154150963, "rewards/judge_quality/mean": 0.39249998331069946, "rewards/judge_quality/std": 0.08892211318016052, "rewards/total_composite/mean": 0.8044524192810059, "rewards/total_composite/std": 0.03227698802947998, "reward": 0.8044524192810059, "reward_std": 0.032276980578899384, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15456882119178772, "sampling/sampling_logp_difference/max": 1.6857328414916992, "sampling/importance_sampling_ratio/min": 0.185308575630188, "sampling/importance_sampling_ratio/mean": 0.9899730682373047, "sampling/importance_sampling_ratio/max": 1.9806053638458252, "entropy": 0.6392244361341, "clip_ratio/low_mean": 0.009615384973585606, "clip_ratio/low_min": 0.009615384973585606, "clip_ratio/high_mean": 0.11469720490276814, "clip_ratio/high_max": 0.11469720490276814, "clip_ratio/region_mean": 0.12431258987635374, "reward_total_mean": 0.8044524192810059, "reward_meter_mean": 0.9814830422401428, "reward_meter_std": 0.02102908305823803, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9503509998321533, "reward_repeat_soft_std": 0.03320911154150963, "reward_judge_quality_mean": 0.39249998331069946, "reward_judge_quality_std": 0.08892211318016052, "reward_total_composite_mean": 0.8044524192810059, "reward_total_composite_std": 0.03227698802947998} {"timestamp_utc": "2026-04-13T02:44:04Z", "mode": "train", "global_step": 1758, "epoch": 0.17659467604218984, "loss": 0.0663, "grad_norm": 14.69132137298584, "learning_rate": 4.675757575757576e-06, "num_tokens": 3185143.0, "completions/mean_length": 32.5, "completions/min_length": 30.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.5, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.8047865629196167, "rewards/meter/std": 0.21322894096374512, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9653917551040649, "rewards/repeat_soft/std": 0.032106392085552216, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7369431257247925, "rewards/total_composite/std": 0.09648601710796356, "reward": 0.7369431257247925, "reward_std": 0.09648600965738297, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11236555129289627, "sampling/sampling_logp_difference/max": 1.4359030723571777, "sampling/importance_sampling_ratio/min": 0.23790042102336884, "sampling/importance_sampling_ratio/mean": 0.9920035600662231, "sampling/importance_sampling_ratio/max": 1.8379878997802734, "entropy": 0.5577161684632301, "clip_ratio/low_mean": 0.029390007257461548, "clip_ratio/low_min": 0.029390007257461548, "clip_ratio/high_mean": 0.05947580561041832, "clip_ratio/high_max": 0.05947580561041832, "clip_ratio/region_mean": 0.08886581286787987, "reward_total_mean": 0.7369431257247925, "reward_meter_mean": 0.8047865629196167, "reward_meter_std": 0.21322894096374512, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9653917551040649, "reward_repeat_soft_std": 0.032106392085552216, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7369431257247925, "reward_total_composite_std": 0.09648601710796356} {"timestamp_utc": "2026-04-13T02:44:11Z", "mode": "train", "global_step": 1759, "epoch": 0.17669512807634355, "loss": 0.0325, "grad_norm": 6.359126567840576, "learning_rate": 4.6727272727272735e-06, "num_tokens": 3187315.0, "completions/mean_length": 100.5, "completions/min_length": 93.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.5, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.7915399074554443, "rewards/meter/std": 0.2663384974002838, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9244115352630615, "rewards/repeat_soft/std": 0.04162711650133133, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.711884081363678, "rewards/total_composite/std": 0.12474881857633591, "reward": 0.711884081363678, "reward_std": 0.12474881857633591, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10358960181474686, "sampling/sampling_logp_difference/max": 2.030205249786377, "sampling/importance_sampling_ratio/min": 0.13130857050418854, "sampling/importance_sampling_ratio/mean": 1.0001815557479858, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5833392702043056, "clip_ratio/low_mean": 0.025580620858818293, "clip_ratio/low_min": 0.025580620858818293, "clip_ratio/high_mean": 0.08477438893169165, "clip_ratio/high_max": 0.08477438893169165, "clip_ratio/region_mean": 0.11035500979050994, "reward_total_mean": 0.711884081363678, "reward_meter_mean": 0.7915399074554443, "reward_meter_std": 0.2663384974002838, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9244115352630615, "reward_repeat_soft_std": 0.04162711650133133, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.711884081363678, "reward_total_composite_std": 0.12474881857633591} {"timestamp_utc": "2026-04-13T02:44:18Z", "mode": "train", "global_step": 1760, "epoch": 0.17679558011049723, "loss": -0.0088, "grad_norm": 13.902256965637207, "learning_rate": 4.66969696969697e-06, "num_tokens": 3188804.0, "completions/mean_length": 29.125, "completions/min_length": 27.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.125, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.8555389642715454, "rewards/meter/std": 0.12289418280124664, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7708537578582764, "rewards/repeat_soft/std": 0.09904000163078308, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.7943279147148132, "rewards/total_composite/std": 0.10256777703762054, "reward": 0.7943279147148132, "reward_std": 0.10256776213645935, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08989141136407852, "sampling/sampling_logp_difference/max": 1.6441547870635986, "sampling/importance_sampling_ratio/min": 0.1931757628917694, "sampling/importance_sampling_ratio/mean": 0.9967292547225952, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3097436800599098, "clip_ratio/low_mean": 0.02986937901005149, "clip_ratio/low_min": 0.02986937901005149, "clip_ratio/high_mean": 0.0272817462682724, "clip_ratio/high_max": 0.0272817462682724, "clip_ratio/region_mean": 0.05715112527832389, "reward_total_mean": 0.7943279147148132, "reward_meter_mean": 0.8555389642715454, "reward_meter_std": 0.12289418280124664, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7708537578582764, "reward_repeat_soft_std": 0.09904000163078308, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.7943279147148132, "reward_total_composite_std": 0.10256777703762054} {"timestamp_utc": "2026-04-13T02:44:25Z", "mode": "train", "global_step": 1761, "epoch": 0.17689603214465094, "loss": 0.0338, "grad_norm": 10.42278003692627, "learning_rate": 4.666666666666667e-06, "num_tokens": 3191225.0, "completions/mean_length": 107.625, "completions/min_length": 105.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.625, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.8118487596511841, "rewards/meter/std": 0.28957873582839966, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9281964302062988, "rewards/repeat_soft/std": 0.054433535784482956, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.752901554107666, "rewards/total_composite/std": 0.14941714704036713, "reward": 0.752901554107666, "reward_std": 0.14941714704036713, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1022859588265419, "sampling/sampling_logp_difference/max": 1.6590697765350342, "sampling/importance_sampling_ratio/min": 0.19031593203544617, "sampling/importance_sampling_ratio/mean": 1.0118815898895264, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5405211746692657, "clip_ratio/low_mean": 0.015840022824704647, "clip_ratio/low_min": 0.015840022824704647, "clip_ratio/high_mean": 0.07633502967655659, "clip_ratio/high_max": 0.07633502967655659, "clip_ratio/region_mean": 0.09217505250126123, "reward_total_mean": 0.752901554107666, "reward_meter_mean": 0.8118487596511841, "reward_meter_std": 0.28957873582839966, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9281964302062988, "reward_repeat_soft_std": 0.054433535784482956, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.752901554107666, "reward_total_composite_std": 0.14941714704036713} {"timestamp_utc": "2026-04-13T02:44:31Z", "mode": "train", "global_step": 1762, "epoch": 0.17699648417880462, "loss": 0.0282, "grad_norm": 12.231925964355469, "learning_rate": 4.663636363636364e-06, "num_tokens": 3192795.0, "completions/mean_length": 46.25, "completions/min_length": 42.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.25, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.7867258787155151, "rewards/meter/std": 0.3128153383731842, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9689304828643799, "rewards/repeat_soft/std": 0.022751672193408012, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.0975411981344223, "rewards/total_composite/mean": 0.7179197072982788, "rewards/total_composite/std": 0.161919504404068, "reward": 0.7179197072982788, "reward_std": 0.1619194895029068, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13072648644447327, "sampling/sampling_logp_difference/max": 1.5135345458984375, "sampling/importance_sampling_ratio/min": 0.2201305329799652, "sampling/importance_sampling_ratio/mean": 0.9976024031639099, "sampling/importance_sampling_ratio/max": 1.7676571607589722, "entropy": 0.7444295361638069, "clip_ratio/low_mean": 0.02543478226289153, "clip_ratio/low_min": 0.02543478226289153, "clip_ratio/high_mean": 0.11537749599665403, "clip_ratio/high_max": 0.11537749599665403, "clip_ratio/region_mean": 0.14081227825954556, "reward_total_mean": 0.7179197072982788, "reward_meter_mean": 0.7867258787155151, "reward_meter_std": 0.3128153383731842, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9689304828643799, "reward_repeat_soft_std": 0.022751672193408012, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.0975411981344223, "reward_total_composite_mean": 0.7179197072982788, "reward_total_composite_std": 0.161919504404068} {"timestamp_utc": "2026-04-13T02:44:37Z", "mode": "train", "global_step": 1763, "epoch": 0.1770969362129583, "loss": 0.0423, "grad_norm": 8.353372573852539, "learning_rate": 4.660606060606061e-06, "num_tokens": 3195024.0, "completions/mean_length": 97.625, "completions/min_length": 93.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.625, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.6716561317443848, "rewards/meter/std": 0.33462682366371155, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9538842439651489, "rewards/repeat_soft/std": 0.03545726463198662, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.6785086989402771, "rewards/total_composite/std": 0.1714162826538086, "reward": 0.6785086989402771, "reward_std": 0.1714162677526474, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1021639034152031, "sampling/sampling_logp_difference/max": 2.3323612213134766, "sampling/importance_sampling_ratio/min": 0.09706628322601318, "sampling/importance_sampling_ratio/mean": 1.0150340795516968, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.605816874653101, "clip_ratio/low_mean": 0.044744846411049366, "clip_ratio/low_min": 0.044744846411049366, "clip_ratio/high_mean": 0.05010634660720825, "clip_ratio/high_max": 0.05010634660720825, "clip_ratio/region_mean": 0.09485119301825762, "reward_total_mean": 0.6785086989402771, "reward_meter_mean": 0.6716561317443848, "reward_meter_std": 0.33462682366371155, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9538842439651489, "reward_repeat_soft_std": 0.03545726463198662, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.6785086989402771, "reward_total_composite_std": 0.1714162826538086} {"timestamp_utc": "2026-04-13T02:44:45Z", "mode": "train", "global_step": 1764, "epoch": 0.177197388247112, "loss": 0.0106, "grad_norm": 13.464056968688965, "learning_rate": 4.657575757575758e-06, "num_tokens": 3196752.0, "completions/mean_length": 48.0, "completions/min_length": 45.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.0, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9842088222503662, "rewards/meter/std": 0.010131052695214748, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.886462926864624, "rewards/repeat_soft/std": 0.1371145248413086, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.8045402765274048, "rewards/total_composite/std": 0.03224100172519684, "reward": 0.8045402765274048, "reward_std": 0.032241009175777435, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11437255144119263, "sampling/sampling_logp_difference/max": 1.9442253112792969, "sampling/importance_sampling_ratio/min": 0.14309804141521454, "sampling/importance_sampling_ratio/mean": 1.0066322088241577, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.567682471126318, "clip_ratio/low_mean": 0.0102777776774019, "clip_ratio/low_min": 0.0102777776774019, "clip_ratio/high_mean": 0.10095855640247464, "clip_ratio/high_max": 0.10095855640247464, "clip_ratio/region_mean": 0.11123633407987654, "reward_total_mean": 0.8045402765274048, "reward_meter_mean": 0.9842088222503662, "reward_meter_std": 0.010131052695214748, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.886462926864624, "reward_repeat_soft_std": 0.1371145248413086, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.8045402765274048, "reward_total_composite_std": 0.03224100172519684} {"timestamp_utc": "2026-04-13T02:44:52Z", "mode": "train", "global_step": 1765, "epoch": 0.1772978402812657, "loss": 0.0121, "grad_norm": 8.614609718322754, "learning_rate": 4.654545454545455e-06, "num_tokens": 3198561.0, "completions/mean_length": 73.125, "completions/min_length": 63.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.125, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9725469946861267, "rewards/meter/std": 0.024202413856983185, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9241282939910889, "rewards/repeat_soft/std": 0.08403926342725754, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8060590028762817, "rewards/total_composite/std": 0.0116598354652524, "reward": 0.8060590028762817, "reward_std": 0.011659827083349228, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10705755650997162, "sampling/sampling_logp_difference/max": 1.8073184490203857, "sampling/importance_sampling_ratio/min": 0.16409358382225037, "sampling/importance_sampling_ratio/mean": 1.0049914121627808, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5455218777060509, "clip_ratio/low_mean": 0.04110549995675683, "clip_ratio/low_min": 0.04110549995675683, "clip_ratio/high_mean": 0.08077302947640419, "clip_ratio/high_max": 0.08077302947640419, "clip_ratio/region_mean": 0.12187852943316102, "reward_total_mean": 0.8060590028762817, "reward_meter_mean": 0.9725469946861267, "reward_meter_std": 0.024202413856983185, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9241282939910889, "reward_repeat_soft_std": 0.08403926342725754, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8060590028762817, "reward_total_composite_std": 0.0116598354652524} {"timestamp_utc": "2026-04-13T02:45:04Z", "mode": "train", "global_step": 1766, "epoch": 0.1773982923154194, "loss": -0.1255, "grad_norm": 2.6900217533111572, "learning_rate": 4.651515151515152e-06, "num_tokens": 3200115.0, "completions/mean_length": 110.25, "completions/min_length": 48.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 52.85714340209961, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.8681445121765137, "rewards/meter/std": 0.20473621785640717, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.747273325920105, "rewards/repeat_soft/std": 0.1147967278957367, "rewards/judge_quality/mean": 0.22999998927116394, "rewards/judge_quality/std": 0.13341663777828217, "rewards/total_composite/mean": 0.6221550107002258, "rewards/total_composite/std": 0.2650856375694275, "reward": 0.6221550107002258, "reward_std": 0.2650856375694275, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0851171612739563, "sampling/sampling_logp_difference/max": 2.3785667419433594, "sampling/importance_sampling_ratio/min": 0.0926833227276802, "sampling/importance_sampling_ratio/mean": 1.0000762939453125, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.26364110223948956, "clip_ratio/low_mean": 0.010080644860863686, "clip_ratio/low_min": 0.010080644860863686, "clip_ratio/high_mean": 0.07610224094241858, "clip_ratio/high_max": 0.07610224094241858, "clip_ratio/region_mean": 0.08618288580328226, "reward_total_mean": 0.6221550107002258, "reward_meter_mean": 0.8681445121765137, "reward_meter_std": 0.20473621785640717, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.747273325920105, "reward_repeat_soft_std": 0.1147967278957367, "reward_judge_quality_mean": 0.22999998927116394, "reward_judge_quality_std": 0.13341663777828217, "reward_total_composite_mean": 0.6221550107002258, "reward_total_composite_std": 0.2650856375694275} {"timestamp_utc": "2026-04-13T02:45:11Z", "mode": "train", "global_step": 1767, "epoch": 0.17749874434957308, "loss": -0.0051, "grad_norm": 15.650751113891602, "learning_rate": 4.648484848484849e-06, "num_tokens": 3201692.0, "completions/mean_length": 34.125, "completions/min_length": 32.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.125, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.8361167907714844, "rewards/meter/std": 0.11810576915740967, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.989013671875, "rewards/repeat_soft/std": 0.010557981207966805, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.7481539249420166, "rewards/total_composite/std": 0.04532990604639053, "reward": 0.7481539249420166, "reward_std": 0.04532989114522934, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10067026317119598, "sampling/sampling_logp_difference/max": 1.9648995399475098, "sampling/importance_sampling_ratio/min": 0.14016996324062347, "sampling/importance_sampling_ratio/mean": 0.9926932454109192, "sampling/importance_sampling_ratio/max": 1.9711856842041016, "entropy": 0.4106884002685547, "clip_ratio/low_mean": 0.025372024159878492, "clip_ratio/low_min": 0.025372024159878492, "clip_ratio/high_mean": 0.05882054683752358, "clip_ratio/high_max": 0.05882054683752358, "clip_ratio/region_mean": 0.08419257099740207, "reward_total_mean": 0.7481539249420166, "reward_meter_mean": 0.8361167907714844, "reward_meter_std": 0.11810576915740967, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.989013671875, "reward_repeat_soft_std": 0.010557981207966805, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.7481539249420166, "reward_total_composite_std": 0.04532990604639053} {"timestamp_utc": "2026-04-13T02:45:17Z", "mode": "train", "global_step": 1768, "epoch": 0.17759919638372676, "loss": 0.0382, "grad_norm": 13.67770767211914, "learning_rate": 4.645454545454545e-06, "num_tokens": 3203593.0, "completions/mean_length": 47.625, "completions/min_length": 44.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.625, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.977152943611145, "rewards/meter/std": 0.014186891727149487, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9302257895469666, "rewards/repeat_soft/std": 0.08689029514789581, "rewards/judge_quality/mean": 0.36374998092651367, "rewards/judge_quality/std": 0.09500939399003983, "rewards/total_composite/mean": 0.7918664216995239, "rewards/total_composite/std": 0.024855386465787888, "reward": 0.7918664216995239, "reward_std": 0.024855399504303932, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1351868212223053, "sampling/sampling_logp_difference/max": 2.691560745239258, "sampling/importance_sampling_ratio/min": 0.06777507811784744, "sampling/importance_sampling_ratio/mean": 0.98212069272995, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5612181723117828, "clip_ratio/low_mean": 0.06032196991145611, "clip_ratio/low_min": 0.06032196991145611, "clip_ratio/high_mean": 0.04687384143471718, "clip_ratio/high_max": 0.04687384143471718, "clip_ratio/region_mean": 0.10719581134617329, "reward_total_mean": 0.7918664216995239, "reward_meter_mean": 0.977152943611145, "reward_meter_std": 0.014186891727149487, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9302257895469666, "reward_repeat_soft_std": 0.08689029514789581, "reward_judge_quality_mean": 0.36374998092651367, "reward_judge_quality_std": 0.09500939399003983, "reward_total_composite_mean": 0.7918664216995239, "reward_total_composite_std": 0.024855386465787888} {"timestamp_utc": "2026-04-13T02:45:24Z", "mode": "train", "global_step": 1769, "epoch": 0.17769964841788047, "loss": 0.0341, "grad_norm": 5.7418694496154785, "learning_rate": 4.642424242424243e-06, "num_tokens": 3206027.0, "completions/mean_length": 124.25, "completions/min_length": 108.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 124.25, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.876604437828064, "rewards/meter/std": 0.12596189975738525, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8586680889129639, "rewards/repeat_soft/std": 0.13247206807136536, "rewards/judge_quality/mean": 0.2762500047683716, "rewards/judge_quality/std": 0.12603145837783813, "rewards/total_composite/mean": 0.7132138013839722, "rewards/total_composite/std": 0.06899500638246536, "reward": 0.7132138013839722, "reward_std": 0.06899500638246536, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09094785153865814, "sampling/sampling_logp_difference/max": 1.2384438514709473, "sampling/importance_sampling_ratio/min": 0.2898348867893219, "sampling/importance_sampling_ratio/mean": 1.012986421585083, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5988048017024994, "clip_ratio/low_mean": 0.03377602808177471, "clip_ratio/low_min": 0.03377602808177471, "clip_ratio/high_mean": 0.0427244221791625, "clip_ratio/high_max": 0.0427244221791625, "clip_ratio/region_mean": 0.07650045026093721, "reward_total_mean": 0.7132138013839722, "reward_meter_mean": 0.876604437828064, "reward_meter_std": 0.12596189975738525, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8586680889129639, "reward_repeat_soft_std": 0.13247206807136536, "reward_judge_quality_mean": 0.2762500047683716, "reward_judge_quality_std": 0.12603145837783813, "reward_total_composite_mean": 0.7132138013839722, "reward_total_composite_std": 0.06899500638246536} {"timestamp_utc": "2026-04-13T02:45:32Z", "mode": "train", "global_step": 1770, "epoch": 0.17780010045203415, "loss": 0.0209, "grad_norm": 7.598031520843506, "learning_rate": 4.63939393939394e-06, "num_tokens": 3208302.0, "completions/mean_length": 102.375, "completions/min_length": 89.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.375, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.8392527103424072, "rewards/meter/std": 0.30464187264442444, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8616254329681396, "rewards/repeat_soft/std": 0.1081162691116333, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7270762324333191, "rewards/total_composite/std": 0.13806019723415375, "reward": 0.7270762324333191, "reward_std": 0.13806016743183136, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09394537657499313, "sampling/sampling_logp_difference/max": 1.4451658725738525, "sampling/importance_sampling_ratio/min": 0.23570698499679565, "sampling/importance_sampling_ratio/mean": 1.0135648250579834, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4796103313565254, "clip_ratio/low_mean": 0.015227645635604858, "clip_ratio/low_min": 0.015227645635604858, "clip_ratio/high_mean": 0.060580164194107056, "clip_ratio/high_max": 0.060580164194107056, "clip_ratio/region_mean": 0.07580780982971191, "reward_total_mean": 0.7270762324333191, "reward_meter_mean": 0.8392527103424072, "reward_meter_std": 0.30464187264442444, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8616254329681396, "reward_repeat_soft_std": 0.1081162691116333, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7270762324333191, "reward_total_composite_std": 0.13806019723415375} {"timestamp_utc": "2026-04-13T02:45:43Z", "mode": "train", "global_step": 1771, "epoch": 0.17790055248618786, "loss": -0.2098, "grad_norm": 1.6846851110458374, "learning_rate": 4.636363636363636e-06, "num_tokens": 3210607.0, "completions/mean_length": 166.125, "completions/min_length": 110.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 116.71429443359375, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.8544009923934937, "rewards/meter/std": 0.3452796936035156, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9002564549446106, "rewards/repeat_soft/std": 0.050611838698387146, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.13715866208076477, "rewards/total_composite/mean": 0.69075608253479, "rewards/total_composite/std": 0.28004124760627747, "reward": 0.69075608253479, "reward_std": 0.2800412178039551, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08955264091491699, "sampling/sampling_logp_difference/max": 1.3894574642181396, "sampling/importance_sampling_ratio/min": 0.24921047687530518, "sampling/importance_sampling_ratio/mean": 1.0036720037460327, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.41169606521725655, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08590755518525839, "clip_ratio/high_max": 0.08590755518525839, "clip_ratio/region_mean": 0.08590755518525839, "reward_total_mean": 0.69075608253479, "reward_meter_mean": 0.8544009923934937, "reward_meter_std": 0.3452796936035156, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9002564549446106, "reward_repeat_soft_std": 0.050611838698387146, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.13715866208076477, "reward_total_composite_mean": 0.69075608253479, "reward_total_composite_std": 0.28004124760627747} {"timestamp_utc": "2026-04-13T02:45:50Z", "mode": "train", "global_step": 1772, "epoch": 0.17800100452034154, "loss": 0.0431, "grad_norm": 7.098455905914307, "learning_rate": 4.633333333333334e-06, "num_tokens": 3213083.0, "completions/mean_length": 117.5, "completions/min_length": 110.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.5, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.1628231704235077, "rewards/meter/std": 0.11915864795446396, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9111521244049072, "rewards/repeat_soft/std": 0.0396176278591156, "rewards/judge_quality/mean": 0.44874998927116394, "rewards/judge_quality/std": 0.16137246787548065, "rewards/total_composite/mean": 0.4490106701850891, "rewards/total_composite/std": 0.0660322904586792, "reward": 0.4490106701850891, "reward_std": 0.0660322979092598, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1129550039768219, "sampling/sampling_logp_difference/max": 1.7169873714447021, "sampling/importance_sampling_ratio/min": 0.17960643768310547, "sampling/importance_sampling_ratio/mean": 0.9963964819908142, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5838475115597248, "clip_ratio/low_mean": 0.05968334339559078, "clip_ratio/low_min": 0.05968334339559078, "clip_ratio/high_mean": 0.04832339845597744, "clip_ratio/high_max": 0.04832339845597744, "clip_ratio/region_mean": 0.10800674185156822, "reward_total_mean": 0.4490106701850891, "reward_meter_mean": 0.1628231704235077, "reward_meter_std": 0.11915864795446396, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9111521244049072, "reward_repeat_soft_std": 0.0396176278591156, "reward_judge_quality_mean": 0.44874998927116394, "reward_judge_quality_std": 0.16137246787548065, "reward_total_composite_mean": 0.4490106701850891, "reward_total_composite_std": 0.0660322904586792} {"timestamp_utc": "2026-04-13T02:45:58Z", "mode": "train", "global_step": 1773, "epoch": 0.17810145655449522, "loss": 0.0849, "grad_norm": 9.924332618713379, "learning_rate": 4.630303030303031e-06, "num_tokens": 3215272.0, "completions/mean_length": 101.625, "completions/min_length": 91.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.625, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.7256331443786621, "rewards/meter/std": 0.3794268071651459, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9548149108886719, "rewards/repeat_soft/std": 0.01588037796318531, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.6916413903236389, "rewards/total_composite/std": 0.18216992914676666, "reward": 0.6916413903236389, "reward_std": 0.18216992914676666, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09846402704715729, "sampling/sampling_logp_difference/max": 1.6784324645996094, "sampling/importance_sampling_ratio/min": 0.1866663545370102, "sampling/importance_sampling_ratio/mean": 1.0110050439834595, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.566569659858942, "clip_ratio/low_mean": 0.04018641263246536, "clip_ratio/low_min": 0.04018641263246536, "clip_ratio/high_mean": 0.05253816209733486, "clip_ratio/high_max": 0.05253816209733486, "clip_ratio/region_mean": 0.09272457472980022, "reward_total_mean": 0.6916413903236389, "reward_meter_mean": 0.7256331443786621, "reward_meter_std": 0.3794268071651459, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9548149108886719, "reward_repeat_soft_std": 0.01588037796318531, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.6916413903236389, "reward_total_composite_std": 0.18216992914676666} {"timestamp_utc": "2026-04-13T02:46:10Z", "mode": "train", "global_step": 1774, "epoch": 0.17820190858864893, "loss": -0.0741, "grad_norm": 3.6104648113250732, "learning_rate": 4.627272727272727e-06, "num_tokens": 3216733.0, "completions/mean_length": 87.625, "completions/min_length": 22.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 27.000001907348633, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.5696398019790649, "rewards/meter/std": 0.39497706294059753, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9624292850494385, "rewards/repeat_soft/std": 0.00019996572518721223, "rewards/judge_quality/mean": 0.3462499976158142, "rewards/judge_quality/std": 0.1487027108669281, "rewards/total_composite/mean": 0.5391632318496704, "rewards/total_composite/std": 0.2716015577316284, "reward": 0.5391632318496704, "reward_std": 0.2716015577316284, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1580711156129837, "sampling/sampling_logp_difference/max": 1.4348104000091553, "sampling/importance_sampling_ratio/min": 0.23816052079200745, "sampling/importance_sampling_ratio/mean": 1.0191214084625244, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7181641981005669, "clip_ratio/low_mean": 0.059976489283144474, "clip_ratio/low_min": 0.059976489283144474, "clip_ratio/high_mean": 0.07752232160419226, "clip_ratio/high_max": 0.07752232160419226, "clip_ratio/region_mean": 0.13749881088733673, "reward_total_mean": 0.5391632318496704, "reward_meter_mean": 0.5696398019790649, "reward_meter_std": 0.39497706294059753, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9624292850494385, "reward_repeat_soft_std": 0.00019996572518721223, "reward_judge_quality_mean": 0.3462499976158142, "reward_judge_quality_std": 0.1487027108669281, "reward_total_composite_mean": 0.5391632318496704, "reward_total_composite_std": 0.2716015577316284} {"timestamp_utc": "2026-04-13T02:46:22Z", "mode": "train", "global_step": 1775, "epoch": 0.1783023606228026, "loss": -0.1375, "grad_norm": 3.3315813541412354, "learning_rate": 4.6242424242424245e-06, "num_tokens": 3218576.0, "completions/mean_length": 122.375, "completions/min_length": 61.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 66.71428680419922, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.7658471465110779, "rewards/meter/std": 0.34032854437828064, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9839047193527222, "rewards/repeat_soft/std": 0.014663629233837128, "rewards/judge_quality/mean": 0.4775000214576721, "rewards/judge_quality/std": 0.2539263069629669, "rewards/total_composite/mean": 0.6473730206489563, "rewards/total_composite/std": 0.302258163690567, "reward": 0.6473730206489563, "reward_std": 0.30225813388824463, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12205831706523895, "sampling/sampling_logp_difference/max": 1.5813627243041992, "sampling/importance_sampling_ratio/min": 0.20569461584091187, "sampling/importance_sampling_ratio/mean": 1.0039063692092896, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5273088701069355, "clip_ratio/low_mean": 0.0234375, "clip_ratio/low_min": 0.0234375, "clip_ratio/high_mean": 0.09308139141649008, "clip_ratio/high_max": 0.09308139141649008, "clip_ratio/region_mean": 0.11651889141649008, "reward_total_mean": 0.6473730206489563, "reward_meter_mean": 0.7658471465110779, "reward_meter_std": 0.34032854437828064, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9839047193527222, "reward_repeat_soft_std": 0.014663629233837128, "reward_judge_quality_mean": 0.4775000214576721, "reward_judge_quality_std": 0.2539263069629669, "reward_total_composite_mean": 0.6473730206489563, "reward_total_composite_std": 0.302258163690567} {"timestamp_utc": "2026-04-13T02:46:28Z", "mode": "train", "global_step": 1776, "epoch": 0.17840281265695632, "loss": 0.0608, "grad_norm": 12.068819999694824, "learning_rate": 4.621212121212122e-06, "num_tokens": 3220220.0, "completions/mean_length": 26.5, "completions/min_length": 22.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.5, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.6702833771705627, "rewards/meter/std": 0.39704373478889465, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8887313008308411, "rewards/repeat_soft/std": 0.05324139446020126, "rewards/judge_quality/mean": 0.39249998331069946, "rewards/judge_quality/std": 0.08892211318016052, "rewards/total_composite/mean": 0.6582506895065308, "rewards/total_composite/std": 0.18664635717868805, "reward": 0.6582506895065308, "reward_std": 0.18664635717868805, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11570510268211365, "sampling/sampling_logp_difference/max": 2.4161009788513184, "sampling/importance_sampling_ratio/min": 0.0892690047621727, "sampling/importance_sampling_ratio/mean": 1.023300290107727, "sampling/importance_sampling_ratio/max": 1.9029524326324463, "entropy": 0.5624456442892551, "clip_ratio/low_mean": 0.027116402983665466, "clip_ratio/low_min": 0.027116402983665466, "clip_ratio/high_mean": 0.060269732028245926, "clip_ratio/high_max": 0.060269732028245926, "clip_ratio/region_mean": 0.08738613501191139, "reward_total_mean": 0.6582506895065308, "reward_meter_mean": 0.6702833771705627, "reward_meter_std": 0.39704373478889465, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8887313008308411, "reward_repeat_soft_std": 0.05324139446020126, "reward_judge_quality_mean": 0.39249998331069946, "reward_judge_quality_std": 0.08892211318016052, "reward_total_composite_mean": 0.6582506895065308, "reward_total_composite_std": 0.18664635717868805} {"timestamp_utc": "2026-04-13T02:46:34Z", "mode": "train", "global_step": 1777, "epoch": 0.17850326469111, "loss": 0.0235, "grad_norm": 9.912286758422852, "learning_rate": 4.618181818181818e-06, "num_tokens": 3221767.0, "completions/mean_length": 31.375, "completions/min_length": 29.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.375, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.850136935710907, "rewards/meter/std": 0.1078629344701767, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9755498766899109, "rewards/repeat_soft/std": 0.02260526269674301, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7572416067123413, "rewards/total_composite/std": 0.04864354431629181, "reward": 0.7572416067123413, "reward_std": 0.048643529415130615, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07958268374204636, "sampling/sampling_logp_difference/max": 1.6978750228881836, "sampling/importance_sampling_ratio/min": 0.18307213485240936, "sampling/importance_sampling_ratio/mean": 1.0061920881271362, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36885247752070427, "clip_ratio/low_mean": 0.01991691067814827, "clip_ratio/low_min": 0.01991691067814827, "clip_ratio/high_mean": 0.03226601053029299, "clip_ratio/high_max": 0.03226601053029299, "clip_ratio/region_mean": 0.05218292120844126, "reward_total_mean": 0.7572416067123413, "reward_meter_mean": 0.850136935710907, "reward_meter_std": 0.1078629344701767, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9755498766899109, "reward_repeat_soft_std": 0.02260526269674301, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7572416067123413, "reward_total_composite_std": 0.04864354431629181} {"timestamp_utc": "2026-04-13T02:46:41Z", "mode": "train", "global_step": 1778, "epoch": 0.17860371672526368, "loss": -0.0025, "grad_norm": 30.87159538269043, "learning_rate": 4.615151515151515e-06, "num_tokens": 3223373.0, "completions/mean_length": 33.75, "completions/min_length": 30.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.75, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.5865691304206848, "rewards/meter/std": 0.43864890933036804, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9822266697883606, "rewards/repeat_soft/std": 0.02038102224469185, "rewards/judge_quality/mean": 0.7112500071525574, "rewards/judge_quality/std": 0.2426895648241043, "rewards/total_composite/mean": 0.7255538105964661, "rewards/total_composite/std": 0.2127108871936798, "reward": 0.7255538105964661, "reward_std": 0.2127108871936798, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11412500590085983, "sampling/sampling_logp_difference/max": 1.6018917560577393, "sampling/importance_sampling_ratio/min": 0.20151492953300476, "sampling/importance_sampling_ratio/mean": 1.0275646448135376, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47950971499085426, "clip_ratio/low_mean": 0.03922822047024965, "clip_ratio/low_min": 0.03922822047024965, "clip_ratio/high_mean": 0.057370311813429, "clip_ratio/high_max": 0.057370311813429, "clip_ratio/region_mean": 0.09659853228367865, "reward_total_mean": 0.7255538105964661, "reward_meter_mean": 0.5865691304206848, "reward_meter_std": 0.43864890933036804, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9822266697883606, "reward_repeat_soft_std": 0.02038102224469185, "reward_judge_quality_mean": 0.7112500071525574, "reward_judge_quality_std": 0.2426895648241043, "reward_total_composite_mean": 0.7255538105964661, "reward_total_composite_std": 0.2127108871936798} {"timestamp_utc": "2026-04-13T02:46:49Z", "mode": "train", "global_step": 1779, "epoch": 0.1787041687594174, "loss": -0.003, "grad_norm": 5.503518104553223, "learning_rate": 4.612121212121212e-06, "num_tokens": 3226338.0, "completions/mean_length": 177.625, "completions/min_length": 160.0, "completions/max_length": 194.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 177.625, "completions/min_terminated_length": 160.0, "completions/max_terminated_length": 194.0, "rewards/meter/mean": 0.4470536410808563, "rewards/meter/std": 0.44793230295181274, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6917934417724609, "rewards/repeat_soft/std": 0.09501318633556366, "rewards/judge_quality/mean": 0.16250000894069672, "rewards/judge_quality/std": 0.0353553369641304, "rewards/total_composite/mean": 0.46285349130630493, "rewards/total_composite/std": 0.20798321068286896, "reward": 0.46285349130630493, "reward_std": 0.20798321068286896, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08848892152309418, "sampling/sampling_logp_difference/max": 1.9978313446044922, "sampling/importance_sampling_ratio/min": 0.1356291025876999, "sampling/importance_sampling_ratio/mean": 1.0085369348526, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5585434138774872, "clip_ratio/low_mean": 0.03333843918517232, "clip_ratio/low_min": 0.03333843918517232, "clip_ratio/high_mean": 0.03716666251420975, "clip_ratio/high_max": 0.03716666251420975, "clip_ratio/region_mean": 0.07050510169938207, "reward_total_mean": 0.46285349130630493, "reward_meter_mean": 0.4470536410808563, "reward_meter_std": 0.44793230295181274, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6917934417724609, "reward_repeat_soft_std": 0.09501318633556366, "reward_judge_quality_mean": 0.16250000894069672, "reward_judge_quality_std": 0.0353553369641304, "reward_total_composite_mean": 0.46285349130630493, "reward_total_composite_std": 0.20798321068286896} {"timestamp_utc": "2026-04-13T02:46:57Z", "mode": "train", "global_step": 1780, "epoch": 0.17880462079357107, "loss": 0.0422, "grad_norm": 10.882675170898438, "learning_rate": 4.60909090909091e-06, "num_tokens": 3228051.0, "completions/mean_length": 64.125, "completions/min_length": 56.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.125, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.7257654666900635, "rewards/meter/std": 0.33096975088119507, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9724231362342834, "rewards/repeat_soft/std": 0.02508735843002796, "rewards/judge_quality/mean": 0.5224999785423279, "rewards/judge_quality/std": 0.17701898515224457, "rewards/total_composite/mean": 0.7118368148803711, "rewards/total_composite/std": 0.17780078947544098, "reward": 0.7118368148803711, "reward_std": 0.17780078947544098, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11858589202165604, "sampling/sampling_logp_difference/max": 2.063685894012451, "sampling/importance_sampling_ratio/min": 0.12698505818843842, "sampling/importance_sampling_ratio/mean": 1.0034282207489014, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5872177854180336, "clip_ratio/low_mean": 0.02651733299717307, "clip_ratio/low_min": 0.02651733299717307, "clip_ratio/high_mean": 0.07027773465961218, "clip_ratio/high_max": 0.07027773465961218, "clip_ratio/region_mean": 0.09679506765678525, "reward_total_mean": 0.7118368148803711, "reward_meter_mean": 0.7257654666900635, "reward_meter_std": 0.33096975088119507, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9724231362342834, "reward_repeat_soft_std": 0.02508735843002796, "reward_judge_quality_mean": 0.5224999785423279, "reward_judge_quality_std": 0.17701898515224457, "reward_total_composite_mean": 0.7118368148803711, "reward_total_composite_std": 0.17780078947544098} {"timestamp_utc": "2026-04-13T02:47:03Z", "mode": "train", "global_step": 1781, "epoch": 0.17890507282772475, "loss": 0.0207, "grad_norm": 10.813126564025879, "learning_rate": 4.606060606060606e-06, "num_tokens": 3229988.0, "completions/mean_length": 71.125, "completions/min_length": 66.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.125, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.871299684047699, "rewards/meter/std": 0.2983897626399994, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8915431499481201, "rewards/repeat_soft/std": 0.11242666840553284, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.7621141672134399, "rewards/total_composite/std": 0.14037513732910156, "reward": 0.7621141672134399, "reward_std": 0.14037513732910156, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10614709556102753, "sampling/sampling_logp_difference/max": 1.4040437936782837, "sampling/importance_sampling_ratio/min": 0.2456018030643463, "sampling/importance_sampling_ratio/mean": 1.0125069618225098, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6909167692065239, "clip_ratio/low_mean": 0.014285714365541935, "clip_ratio/low_min": 0.014285714365541935, "clip_ratio/high_mean": 0.0926412888802588, "clip_ratio/high_max": 0.0926412888802588, "clip_ratio/region_mean": 0.10692700324580073, "reward_total_mean": 0.7621141672134399, "reward_meter_mean": 0.871299684047699, "reward_meter_std": 0.2983897626399994, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8915431499481201, "reward_repeat_soft_std": 0.11242666840553284, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.7621141672134399, "reward_total_composite_std": 0.14037513732910156} {"timestamp_utc": "2026-04-13T02:47:15Z", "mode": "train", "global_step": 1782, "epoch": 0.17900552486187846, "loss": -0.08, "grad_norm": 2.4788339138031006, "learning_rate": 4.603030303030304e-06, "num_tokens": 3231454.0, "completions/mean_length": 87.25, "completions/min_length": 21.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 26.571430206298828, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.6786673069000244, "rewards/meter/std": 0.42057323455810547, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.957673192024231, "rewards/repeat_soft/std": 0.013428457081317902, "rewards/judge_quality/mean": 0.2800000011920929, "rewards/judge_quality/std": 0.14172407984733582, "rewards/total_composite/mean": 0.6022560596466064, "rewards/total_composite/std": 0.27512863278388977, "reward": 0.6022560596466064, "reward_std": 0.27512863278388977, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12670239806175232, "sampling/sampling_logp_difference/max": 0.8310021162033081, "sampling/importance_sampling_ratio/min": 0.455229789018631, "sampling/importance_sampling_ratio/mean": 1.0298542976379395, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6841128021478653, "clip_ratio/low_mean": 0.020525451749563217, "clip_ratio/low_min": 0.020525451749563217, "clip_ratio/high_mean": 0.0828881780616939, "clip_ratio/high_max": 0.0828881780616939, "clip_ratio/region_mean": 0.10341362981125712, "reward_total_mean": 0.6022560596466064, "reward_meter_mean": 0.6786673069000244, "reward_meter_std": 0.42057323455810547, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.957673192024231, "reward_repeat_soft_std": 0.013428457081317902, "reward_judge_quality_mean": 0.2800000011920929, "reward_judge_quality_std": 0.14172407984733582, "reward_total_composite_mean": 0.6022560596466064, "reward_total_composite_std": 0.27512863278388977} {"timestamp_utc": "2026-04-13T02:47:22Z", "mode": "train", "global_step": 1783, "epoch": 0.17910597689603214, "loss": 0.0031, "grad_norm": 18.007633209228516, "learning_rate": 4.600000000000001e-06, "num_tokens": 3233149.0, "completions/mean_length": 46.875, "completions/min_length": 40.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.875, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.6446491479873657, "rewards/meter/std": 0.3846195936203003, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9766465425491333, "rewards/repeat_soft/std": 0.029938656836748123, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.6573817729949951, "rewards/total_composite/std": 0.166500523686409, "reward": 0.6573817729949951, "reward_std": 0.1665005087852478, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13832499086856842, "sampling/sampling_logp_difference/max": 1.8172845840454102, "sampling/importance_sampling_ratio/min": 0.16246630251407623, "sampling/importance_sampling_ratio/mean": 0.9899571537971497, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8505435958504677, "clip_ratio/low_mean": 0.039860372664406896, "clip_ratio/low_min": 0.039860372664406896, "clip_ratio/high_mean": 0.0738029913045466, "clip_ratio/high_max": 0.0738029913045466, "clip_ratio/region_mean": 0.11366336396895349, "reward_total_mean": 0.6573817729949951, "reward_meter_mean": 0.6446491479873657, "reward_meter_std": 0.3846195936203003, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9766465425491333, "reward_repeat_soft_std": 0.029938656836748123, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.6573817729949951, "reward_total_composite_std": 0.166500523686409} {"timestamp_utc": "2026-04-13T02:47:29Z", "mode": "train", "global_step": 1784, "epoch": 0.17920642893018585, "loss": -0.0086, "grad_norm": 9.826176643371582, "learning_rate": 4.596969696969697e-06, "num_tokens": 3234900.0, "completions/mean_length": 53.875, "completions/min_length": 47.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.875, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9758856296539307, "rewards/meter/std": 0.026138950139284134, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9376334547996521, "rewards/repeat_soft/std": 0.05581635236740112, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.18873640894889832, "rewards/total_composite/mean": 0.8411619663238525, "rewards/total_composite/std": 0.06233620271086693, "reward": 0.8411619663238525, "reward_std": 0.06233619153499603, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10909334570169449, "sampling/sampling_logp_difference/max": 1.5550861358642578, "sampling/importance_sampling_ratio/min": 0.21117118000984192, "sampling/importance_sampling_ratio/mean": 1.0009211301803589, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7255359217524529, "clip_ratio/low_mean": 0.07555951317772269, "clip_ratio/low_min": 0.07555951317772269, "clip_ratio/high_mean": 0.033333334140479565, "clip_ratio/high_max": 0.033333334140479565, "clip_ratio/region_mean": 0.10889284731820226, "reward_total_mean": 0.8411619663238525, "reward_meter_mean": 0.9758856296539307, "reward_meter_std": 0.026138950139284134, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9376334547996521, "reward_repeat_soft_std": 0.05581635236740112, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.18873640894889832, "reward_total_composite_mean": 0.8411619663238525, "reward_total_composite_std": 0.06233620271086693} {"timestamp_utc": "2026-04-13T02:47:36Z", "mode": "train", "global_step": 1785, "epoch": 0.17930688096433953, "loss": 0.0135, "grad_norm": 10.874298095703125, "learning_rate": 4.5939393939393945e-06, "num_tokens": 3237542.0, "completions/mean_length": 138.25, "completions/min_length": 130.0, "completions/max_length": 147.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 138.25, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 147.0, "rewards/meter/mean": 0.9873032569885254, "rewards/meter/std": 0.008256982080638409, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8418489694595337, "rewards/repeat_soft/std": 0.07363144308328629, "rewards/judge_quality/mean": 0.23374998569488525, "rewards/judge_quality/std": 0.09006941318511963, "rewards/total_composite/mean": 0.7485963702201843, "rewards/total_composite/std": 0.031073380261659622, "reward": 0.7485963702201843, "reward_std": 0.031073393300175667, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07884439826011658, "sampling/sampling_logp_difference/max": 1.6450072526931763, "sampling/importance_sampling_ratio/min": 0.19301116466522217, "sampling/importance_sampling_ratio/mean": 1.0081613063812256, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3926781378686428, "clip_ratio/low_mean": 0.021909488830715418, "clip_ratio/low_min": 0.021909488830715418, "clip_ratio/high_mean": 0.0382009306922555, "clip_ratio/high_max": 0.0382009306922555, "clip_ratio/region_mean": 0.060110419522970915, "reward_total_mean": 0.7485963702201843, "reward_meter_mean": 0.9873032569885254, "reward_meter_std": 0.008256982080638409, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8418489694595337, "reward_repeat_soft_std": 0.07363144308328629, "reward_judge_quality_mean": 0.23374998569488525, "reward_judge_quality_std": 0.09006941318511963, "reward_total_composite_mean": 0.7485963702201843, "reward_total_composite_std": 0.031073380261659622} {"timestamp_utc": "2026-04-13T02:47:42Z", "mode": "train", "global_step": 1786, "epoch": 0.1794073329984932, "loss": 0.0114, "grad_norm": 10.626298904418945, "learning_rate": 4.590909090909092e-06, "num_tokens": 3239212.0, "completions/mean_length": 53.75, "completions/min_length": 52.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.75, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9376751184463501, "rewards/meter/std": 0.09882613271474838, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9715579748153687, "rewards/repeat_soft/std": 0.029476825147867203, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.8288595676422119, "rewards/total_composite/std": 0.05050000920891762, "reward": 0.8288595676422119, "reward_std": 0.05050002411007881, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13070517778396606, "sampling/sampling_logp_difference/max": 2.11574125289917, "sampling/importance_sampling_ratio/min": 0.12054390460252762, "sampling/importance_sampling_ratio/mean": 1.0119199752807617, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8400430753827095, "clip_ratio/low_mean": 0.06793519016355276, "clip_ratio/low_min": 0.06793519016355276, "clip_ratio/high_mean": 0.0346840675920248, "clip_ratio/high_max": 0.0346840675920248, "clip_ratio/region_mean": 0.10261925775557756, "reward_total_mean": 0.8288595676422119, "reward_meter_mean": 0.9376751184463501, "reward_meter_std": 0.09882613271474838, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9715579748153687, "reward_repeat_soft_std": 0.029476825147867203, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.8288595676422119, "reward_total_composite_std": 0.05050000920891762} {"timestamp_utc": "2026-04-13T02:47:49Z", "mode": "train", "global_step": 1787, "epoch": 0.17950778503264692, "loss": 0.017, "grad_norm": 12.46412181854248, "learning_rate": 4.587878787878788e-06, "num_tokens": 3241359.0, "completions/mean_length": 88.375, "completions/min_length": 80.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 88.375, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.7096343040466309, "rewards/meter/std": 0.2628750205039978, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8347210884094238, "rewards/repeat_soft/std": 0.11081143468618393, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.6027240753173828, "rewards/total_composite/std": 0.2586403787136078, "reward": 0.6027240753173828, "reward_std": 0.2586403787136078, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09179135411977768, "sampling/sampling_logp_difference/max": 1.7044806480407715, "sampling/importance_sampling_ratio/min": 0.1818668097257614, "sampling/importance_sampling_ratio/mean": 1.004288911819458, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4793449230492115, "clip_ratio/low_mean": 0.021635258104652166, "clip_ratio/low_min": 0.021635258104652166, "clip_ratio/high_mean": 0.06657458981499076, "clip_ratio/high_max": 0.06657458981499076, "clip_ratio/region_mean": 0.08820984791964293, "reward_total_mean": 0.6027240753173828, "reward_meter_mean": 0.7096343040466309, "reward_meter_std": 0.2628750205039978, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8347210884094238, "reward_repeat_soft_std": 0.11081143468618393, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.6027240753173828, "reward_total_composite_std": 0.2586403787136078} {"timestamp_utc": "2026-04-13T02:48:01Z", "mode": "train", "global_step": 1788, "epoch": 0.1796082370668006, "loss": -0.2019, "grad_norm": 2.3512864112854004, "learning_rate": 4.5848484848484854e-06, "num_tokens": 3243707.0, "completions/mean_length": 161.5, "completions/min_length": 90.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 111.42857360839844, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.7946658134460449, "rewards/meter/std": 0.23419535160064697, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9166491031646729, "rewards/repeat_soft/std": 0.03919072076678276, "rewards/judge_quality/mean": 0.33124998211860657, "rewards/judge_quality/std": 0.13715866208076477, "rewards/total_composite/mean": 0.6230775117874146, "rewards/total_composite/std": 0.2782657742500305, "reward": 0.6230775117874146, "reward_std": 0.2782657742500305, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10257165879011154, "sampling/sampling_logp_difference/max": 1.470015525817871, "sampling/importance_sampling_ratio/min": 0.22992190718650818, "sampling/importance_sampling_ratio/mean": 1.0074236392974854, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5263826623558998, "clip_ratio/low_mean": 0.03637853777036071, "clip_ratio/low_min": 0.03637853777036071, "clip_ratio/high_mean": 0.047615486197173595, "clip_ratio/high_max": 0.047615486197173595, "clip_ratio/region_mean": 0.0839940239675343, "reward_total_mean": 0.6230775117874146, "reward_meter_mean": 0.7946658134460449, "reward_meter_std": 0.23419535160064697, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9166491031646729, "reward_repeat_soft_std": 0.03919072076678276, "reward_judge_quality_mean": 0.33124998211860657, "reward_judge_quality_std": 0.13715866208076477, "reward_total_composite_mean": 0.6230775117874146, "reward_total_composite_std": 0.2782657742500305} {"timestamp_utc": "2026-04-13T02:48:08Z", "mode": "train", "global_step": 1789, "epoch": 0.1797086891009543, "loss": 0.0302, "grad_norm": 7.126922130584717, "learning_rate": 4.581818181818183e-06, "num_tokens": 3245893.0, "completions/mean_length": 98.25, "completions/min_length": 91.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.25, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.97145015001297, "rewards/meter/std": 0.024578185752034187, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9270191192626953, "rewards/repeat_soft/std": 0.05685467645525932, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.8107295036315918, "rewards/total_composite/std": 0.03386649116873741, "reward": 0.8107295036315918, "reward_std": 0.033866506069898605, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12786410748958588, "sampling/sampling_logp_difference/max": 2.3485536575317383, "sampling/importance_sampling_ratio/min": 0.0955071970820427, "sampling/importance_sampling_ratio/mean": 1.0087358951568604, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6180082932114601, "clip_ratio/low_mean": 0.06772516202181578, "clip_ratio/low_min": 0.06772516202181578, "clip_ratio/high_mean": 0.05176326632499695, "clip_ratio/high_max": 0.05176326632499695, "clip_ratio/region_mean": 0.11948842834681273, "reward_total_mean": 0.8107295036315918, "reward_meter_mean": 0.97145015001297, "reward_meter_std": 0.024578185752034187, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9270191192626953, "reward_repeat_soft_std": 0.05685467645525932, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.8107295036315918, "reward_total_composite_std": 0.03386649116873741} {"timestamp_utc": "2026-04-13T02:48:14Z", "mode": "train", "global_step": 1790, "epoch": 0.17980914113510799, "loss": 0.0145, "grad_norm": 17.76902198791504, "learning_rate": 4.578787878787879e-06, "num_tokens": 3247387.0, "completions/mean_length": 27.75, "completions/min_length": 24.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.75, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.585739016532898, "rewards/meter/std": 0.4220087230205536, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9514135122299194, "rewards/repeat_soft/std": 0.018294211477041245, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.7097238898277283, "rewards/total_composite/std": 0.2338714897632599, "reward": 0.7097238898277283, "reward_std": 0.23387150466442108, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12703725695610046, "sampling/sampling_logp_difference/max": 1.6101126670837402, "sampling/importance_sampling_ratio/min": 0.19986510276794434, "sampling/importance_sampling_ratio/mean": 1.0051755905151367, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.678428802639246, "clip_ratio/low_mean": 0.06377801764756441, "clip_ratio/low_min": 0.06377801764756441, "clip_ratio/high_mean": 0.05352510418742895, "clip_ratio/high_max": 0.05352510418742895, "clip_ratio/region_mean": 0.11730312183499336, "reward_total_mean": 0.7097238898277283, "reward_meter_mean": 0.585739016532898, "reward_meter_std": 0.4220087230205536, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9514135122299194, "reward_repeat_soft_std": 0.018294211477041245, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.7097238898277283, "reward_total_composite_std": 0.2338714897632599} {"timestamp_utc": "2026-04-13T02:48:22Z", "mode": "train", "global_step": 1791, "epoch": 0.17990959316926167, "loss": 0.0455, "grad_norm": 5.760233402252197, "learning_rate": 4.575757575757576e-06, "num_tokens": 3250175.0, "completions/mean_length": 139.5, "completions/min_length": 126.0, "completions/max_length": 155.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 139.5, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 155.0, "rewards/meter/mean": 0.9801784753799438, "rewards/meter/std": 0.014529249630868435, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7334439158439636, "rewards/repeat_soft/std": 0.08615828305482864, "rewards/judge_quality/mean": 0.30124998092651367, "rewards/judge_quality/std": 0.10398317128419876, "rewards/total_composite/mean": 0.7547997236251831, "rewards/total_composite/std": 0.037760693579912186, "reward": 0.7547997236251831, "reward_std": 0.03776070103049278, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08065392076969147, "sampling/sampling_logp_difference/max": 2.114722728729248, "sampling/importance_sampling_ratio/min": 0.1206667348742485, "sampling/importance_sampling_ratio/mean": 1.0097105503082275, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40758711099624634, "clip_ratio/low_mean": 0.04279692517593503, "clip_ratio/low_min": 0.04279692517593503, "clip_ratio/high_mean": 0.02873964887112379, "clip_ratio/high_max": 0.02873964887112379, "clip_ratio/region_mean": 0.07153657404705882, "reward_total_mean": 0.7547997236251831, "reward_meter_mean": 0.9801784753799438, "reward_meter_std": 0.014529249630868435, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7334439158439636, "reward_repeat_soft_std": 0.08615828305482864, "reward_judge_quality_mean": 0.30124998092651367, "reward_judge_quality_std": 0.10398317128419876, "reward_total_composite_mean": 0.7547997236251831, "reward_total_composite_std": 0.037760693579912186} {"timestamp_utc": "2026-04-13T02:48:29Z", "mode": "train", "global_step": 1792, "epoch": 0.18001004520341538, "loss": -0.0174, "grad_norm": 7.774007797241211, "learning_rate": 4.572727272727273e-06, "num_tokens": 3251934.0, "completions/mean_length": 57.875, "completions/min_length": 49.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.875, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.8059428334236145, "rewards/meter/std": 0.337230920791626, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9385069012641907, "rewards/repeat_soft/std": 0.04053356125950813, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7381500005722046, "rewards/total_composite/std": 0.1517692506313324, "reward": 0.7381500005722046, "reward_std": 0.1517692506313324, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0993642657995224, "sampling/sampling_logp_difference/max": 1.311697006225586, "sampling/importance_sampling_ratio/min": 0.26936256885528564, "sampling/importance_sampling_ratio/mean": 1.0163350105285645, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7405005469918251, "clip_ratio/low_mean": 0.024966930970549583, "clip_ratio/low_min": 0.024966930970549583, "clip_ratio/high_mean": 0.07215288560837507, "clip_ratio/high_max": 0.07215288560837507, "clip_ratio/region_mean": 0.09711981657892466, "reward_total_mean": 0.7381500005722046, "reward_meter_mean": 0.8059428334236145, "reward_meter_std": 0.337230920791626, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9385069012641907, "reward_repeat_soft_std": 0.04053356125950813, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7381500005722046, "reward_total_composite_std": 0.1517692506313324} {"timestamp_utc": "2026-04-13T02:48:35Z", "mode": "train", "global_step": 1793, "epoch": 0.18011049723756906, "loss": -0.0336, "grad_norm": 11.746275901794434, "learning_rate": 4.56969696969697e-06, "num_tokens": 3253408.0, "completions/mean_length": 34.25, "completions/min_length": 31.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.25, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.6580509543418884, "rewards/meter/std": 0.3318418264389038, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9789660573005676, "rewards/repeat_soft/std": 0.01146512571722269, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.6711444854736328, "rewards/total_composite/std": 0.14997495710849762, "reward": 0.6711444854736328, "reward_std": 0.14997494220733643, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11111447215080261, "sampling/sampling_logp_difference/max": 1.6088895797729492, "sampling/importance_sampling_ratio/min": 0.20010969042778015, "sampling/importance_sampling_ratio/mean": 0.9900774359703064, "sampling/importance_sampling_ratio/max": 1.8617016077041626, "entropy": 0.531346756964922, "clip_ratio/low_mean": 0.023452773690223694, "clip_ratio/low_min": 0.023452773690223694, "clip_ratio/high_mean": 0.09722577407956123, "clip_ratio/high_max": 0.09722577407956123, "clip_ratio/region_mean": 0.12067854776978493, "reward_total_mean": 0.6711444854736328, "reward_meter_mean": 0.6580509543418884, "reward_meter_std": 0.3318418264389038, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9789660573005676, "reward_repeat_soft_std": 0.01146512571722269, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.6711444854736328, "reward_total_composite_std": 0.14997495710849762} {"timestamp_utc": "2026-04-13T02:48:47Z", "mode": "train", "global_step": 1794, "epoch": 0.18021094927172276, "loss": 0.0248, "grad_norm": 9.342144966125488, "learning_rate": 4.566666666666667e-06, "num_tokens": 3255148.0, "completions/mean_length": 57.5, "completions/min_length": 54.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.5, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.8112075328826904, "rewards/meter/std": 0.27867424488067627, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9607177376747131, "rewards/repeat_soft/std": 0.02351691573858261, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7382401823997498, "rewards/total_composite/std": 0.12532949447631836, "reward": 0.7382401823997498, "reward_std": 0.12532949447631836, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11585400998592377, "sampling/sampling_logp_difference/max": 1.6180171966552734, "sampling/importance_sampling_ratio/min": 0.19829148054122925, "sampling/importance_sampling_ratio/mean": 1.0140511989593506, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7771035134792328, "clip_ratio/low_mean": 0.03375706262886524, "clip_ratio/low_min": 0.03375706262886524, "clip_ratio/high_mean": 0.075560811907053, "clip_ratio/high_max": 0.075560811907053, "clip_ratio/region_mean": 0.10931787453591824, "reward_total_mean": 0.7382401823997498, "reward_meter_mean": 0.8112075328826904, "reward_meter_std": 0.27867424488067627, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9607177376747131, "reward_repeat_soft_std": 0.02351691573858261, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7382401823997498, "reward_total_composite_std": 0.12532949447631836} {"timestamp_utc": "2026-04-13T02:49:04Z", "mode": "train", "global_step": 1795, "epoch": 0.18031140130587645, "loss": -0.0025, "grad_norm": 8.56093692779541, "learning_rate": 4.563636363636364e-06, "num_tokens": 3257026.0, "completions/mean_length": 79.75, "completions/min_length": 71.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.75, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.8113051652908325, "rewards/meter/std": 0.2664308249950409, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8600702285766602, "rewards/repeat_soft/std": 0.11772489547729492, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.7495943307876587, "rewards/total_composite/std": 0.12656298279762268, "reward": 0.7495943307876587, "reward_std": 0.1265629678964615, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11417996138334274, "sampling/sampling_logp_difference/max": 1.960953712463379, "sampling/importance_sampling_ratio/min": 0.14072413742542267, "sampling/importance_sampling_ratio/mean": 1.0025510787963867, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6334204934537411, "clip_ratio/low_mean": 0.025180180557072163, "clip_ratio/low_min": 0.025180180557072163, "clip_ratio/high_mean": 0.08875122014433146, "clip_ratio/high_max": 0.08875122014433146, "clip_ratio/region_mean": 0.11393140070140362, "reward_total_mean": 0.7495943307876587, "reward_meter_mean": 0.8113051652908325, "reward_meter_std": 0.2664308249950409, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8600702285766602, "reward_repeat_soft_std": 0.11772489547729492, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.7495943307876587, "reward_total_composite_std": 0.12656298279762268} {"timestamp_utc": "2026-04-13T02:49:20Z", "mode": "train", "global_step": 1796, "epoch": 0.18041185334003013, "loss": -0.024, "grad_norm": 5.647561550140381, "learning_rate": 4.560606060606061e-06, "num_tokens": 3259061.0, "completions/mean_length": 89.375, "completions/min_length": 81.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.375, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.9829541444778442, "rewards/meter/std": 0.018602712079882622, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9251039028167725, "rewards/repeat_soft/std": 0.04121231660246849, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.8007147312164307, "rewards/total_composite/std": 0.03018900938332081, "reward": 0.8007147312164307, "reward_std": 0.030189018696546555, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0832136943936348, "sampling/sampling_logp_difference/max": 1.7530264854431152, "sampling/importance_sampling_ratio/min": 0.17324881255626678, "sampling/importance_sampling_ratio/mean": 1.0075604915618896, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4809896871447563, "clip_ratio/low_mean": 0.02568342164158821, "clip_ratio/low_min": 0.02568342164158821, "clip_ratio/high_mean": 0.05828523822128773, "clip_ratio/high_max": 0.05828523822128773, "clip_ratio/region_mean": 0.08396865986287594, "reward_total_mean": 0.8007147312164307, "reward_meter_mean": 0.9829541444778442, "reward_meter_std": 0.018602712079882622, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9251039028167725, "reward_repeat_soft_std": 0.04121231660246849, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.8007147312164307, "reward_total_composite_std": 0.03018900938332081} {"timestamp_utc": "2026-04-13T02:49:36Z", "mode": "train", "global_step": 1797, "epoch": 0.18051230537418383, "loss": -0.2065, "grad_norm": 2.0047314167022705, "learning_rate": 4.557575757575758e-06, "num_tokens": 3261239.0, "completions/mean_length": 337.25, "completions/min_length": 144.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.5, "completions/mean_terminated_length": 162.5, "completions/min_terminated_length": 144.0, "completions/max_terminated_length": 177.0, "rewards/meter/mean": 0.529293417930603, "rewards/meter/std": 0.3499910533428192, "rewards/count_adherence/mean": 0.7749999761581421, "rewards/count_adherence/std": 0.29154759645462036, "rewards/hard_gate/mean": 0.5, "rewards/hard_gate/std": 0.5345224738121033, "rewards/repeat_soft/mean": 0.9059320688247681, "rewards/repeat_soft/std": 0.10445018857717514, "rewards/judge_quality/mean": 0.14000000059604645, "rewards/judge_quality/std": 0.13866712152957916, "rewards/total_composite/mean": 0.28456398844718933, "rewards/total_composite/std": 0.3249357342720032, "reward": 0.28456398844718933, "reward_std": 0.32493576407432556, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1282980889081955, "sampling/sampling_logp_difference/max": 1.535201072692871, "sampling/importance_sampling_ratio/min": 0.21541237831115723, "sampling/importance_sampling_ratio/mean": 1.0256808996200562, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5380111336708069, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.05039241351187229, "clip_ratio/high_max": 0.05039241351187229, "clip_ratio/region_mean": 0.05039241351187229, "reward_total_mean": 0.28456398844718933, "reward_meter_mean": 0.529293417930603, "reward_meter_std": 0.3499910533428192, "reward_count_adherence_mean": 0.7749999761581421, "reward_count_adherence_std": 0.29154759645462036, "reward_hard_gate_mean": 0.5, "reward_hard_gate_std": 0.5345224738121033, "reward_repeat_soft_mean": 0.9059320688247681, "reward_repeat_soft_std": 0.10445018857717514, "reward_judge_quality_mean": 0.14000000059604645, "reward_judge_quality_std": 0.13866712152957916, "reward_total_composite_mean": 0.28456398844718933, "reward_total_composite_std": 0.3249357342720032} {"timestamp_utc": "2026-04-13T02:49:46Z", "mode": "train", "global_step": 1798, "epoch": 0.18061275740833752, "loss": 0.027, "grad_norm": 13.010977745056152, "learning_rate": 4.554545454545455e-06, "num_tokens": 3262861.0, "completions/mean_length": 52.75, "completions/min_length": 45.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.75, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.7893592119216919, "rewards/meter/std": 0.3438335359096527, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8857582807540894, "rewards/repeat_soft/std": 0.13996802270412445, "rewards/judge_quality/mean": 0.41749998927116394, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.7190374732017517, "rewards/total_composite/std": 0.14575138688087463, "reward": 0.7190374732017517, "reward_std": 0.14575137197971344, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11891786009073257, "sampling/sampling_logp_difference/max": 1.8336257934570312, "sampling/importance_sampling_ratio/min": 0.15983299911022186, "sampling/importance_sampling_ratio/mean": 1.0215213298797607, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6393256112933159, "clip_ratio/low_mean": 0.018477043602615595, "clip_ratio/low_min": 0.018477043602615595, "clip_ratio/high_mean": 0.08846338419243693, "clip_ratio/high_max": 0.08846338419243693, "clip_ratio/region_mean": 0.10694042779505253, "reward_total_mean": 0.7190374732017517, "reward_meter_mean": 0.7893592119216919, "reward_meter_std": 0.3438335359096527, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8857582807540894, "reward_repeat_soft_std": 0.13996802270412445, "reward_judge_quality_mean": 0.41749998927116394, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.7190374732017517, "reward_total_composite_std": 0.14575138688087463} {"timestamp_utc": "2026-04-13T02:50:00Z", "mode": "train", "global_step": 1799, "epoch": 0.18071320944249122, "loss": 0.0002, "grad_norm": 5.932833194732666, "learning_rate": 4.551515151515152e-06, "num_tokens": 3265202.0, "completions/mean_length": 127.625, "completions/min_length": 115.0, "completions/max_length": 142.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.625, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 142.0, "rewards/meter/mean": 0.8737084865570068, "rewards/meter/std": 0.3409660756587982, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8297699093818665, "rewards/repeat_soft/std": 0.1081731766462326, "rewards/judge_quality/mean": 0.2487500011920929, "rewards/judge_quality/std": 0.1196945458650589, "rewards/total_composite/mean": 0.7007707953453064, "rewards/total_composite/std": 0.15905292332172394, "reward": 0.7007707953453064, "reward_std": 0.15905292332172394, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09738950431346893, "sampling/sampling_logp_difference/max": 1.7428662776947021, "sampling/importance_sampling_ratio/min": 0.17501802742481232, "sampling/importance_sampling_ratio/mean": 1.004564642906189, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5659884251654148, "clip_ratio/low_mean": 0.011956521309912205, "clip_ratio/low_min": 0.011956521309912205, "clip_ratio/high_mean": 0.07710392586886883, "clip_ratio/high_max": 0.07710392586886883, "clip_ratio/region_mean": 0.08906044717878103, "reward_total_mean": 0.7007707953453064, "reward_meter_mean": 0.8737084865570068, "reward_meter_std": 0.3409660756587982, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8297699093818665, "reward_repeat_soft_std": 0.1081731766462326, "reward_judge_quality_mean": 0.2487500011920929, "reward_judge_quality_std": 0.1196945458650589, "reward_total_composite_mean": 0.7007707953453064, "reward_total_composite_std": 0.15905292332172394} {"timestamp_utc": "2026-04-13T02:50:11Z", "mode": "train", "global_step": 1800, "epoch": 0.1808136614766449, "loss": 0.0493, "grad_norm": 4.5673699378967285, "learning_rate": 4.548484848484849e-06, "num_tokens": 3268098.0, "completions/mean_length": 178.0, "completions/min_length": 144.0, "completions/max_length": 194.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 178.0, "completions/min_terminated_length": 144.0, "completions/max_terminated_length": 194.0, "rewards/meter/mean": 0.8507214784622192, "rewards/meter/std": 0.1618632674217224, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8845824599266052, "rewards/repeat_soft/std": 0.04221129044890404, "rewards/judge_quality/mean": 0.26375001668930054, "rewards/judge_quality/std": 0.13373079895973206, "rewards/total_composite/mean": 0.6854078769683838, "rewards/total_composite/std": 0.09420323371887207, "reward": 0.6854078769683838, "reward_std": 0.09420324116945267, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09060268104076385, "sampling/sampling_logp_difference/max": 1.266718864440918, "sampling/importance_sampling_ratio/min": 0.28175458312034607, "sampling/importance_sampling_ratio/mean": 1.0228074789047241, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.659310020506382, "clip_ratio/low_mean": 0.03613903233781457, "clip_ratio/low_min": 0.03613903233781457, "clip_ratio/high_mean": 0.04480411438271403, "clip_ratio/high_max": 0.04480411438271403, "clip_ratio/region_mean": 0.0809431467205286, "reward_total_mean": 0.6854078769683838, "reward_meter_mean": 0.8507214784622192, "reward_meter_std": 0.1618632674217224, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8845824599266052, "reward_repeat_soft_std": 0.04221129044890404, "reward_judge_quality_mean": 0.26375001668930054, "reward_judge_quality_std": 0.13373079895973206, "reward_total_composite_mean": 0.6854078769683838, "reward_total_composite_std": 0.09420323371887207} {"timestamp_utc": "2026-04-13T02:51:20Z", "mode": "eval", "global_step": 1800, "epoch": 0.1808136614766449, "eval_loss": NaN, "eval_runtime": 67.9567, "eval_samples_per_second": 1.177, "eval_steps_per_second": 0.147, "eval_num_tokens": 3268098.0, "eval_completions/mean_length": 90.575, "eval_completions/min_length": 38.0, "eval_completions/max_length": 187.2, "eval_completions/clipped_ratio": 0.0125, "eval_completions/mean_terminated_length": 85.27857208251953, "eval_completions/min_terminated_length": 38.0, "eval_completions/max_terminated_length": 154.6, "eval_rewards/meter/mean": 0.8212613642215729, "eval_rewards/meter/std": 0.2558968536555767, "eval_rewards/count_adherence/mean": 0.9854166626930236, "eval_rewards/count_adherence/std": 0.03405027724802494, "eval_rewards/hard_gate/mean": 0.9875, "eval_rewards/hard_gate/std": 0.03535533845424652, "eval_rewards/repeat_soft/mean": 0.9022143065929413, "eval_rewards/repeat_soft/std": 0.07740956265479326, "eval_rewards/judge_quality/mean": 0.40862499475479125, "eval_rewards/judge_quality/std": 0.15207917941734195, "eval_rewards/total_composite/mean": 0.7228510618209839, "eval_rewards/total_composite/std": 0.15152148604393006, "eval_reward": 0.7228510618209839, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.052783616632223126, "eval_sampling/sampling_logp_difference/max": 0.8617889881134033, "eval_sampling/importance_sampling_ratio/min": 0.429345703125, "eval_sampling/importance_sampling_ratio/mean": 1.0132127046585082, "eval_sampling/importance_sampling_ratio/max": 1.3670085668563843, "eval_entropy": 0.5594794183969498, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7228510618209839, "eval_reward_meter_mean": 0.8212613642215729, "eval_reward_meter_std": 0.2558968536555767, "eval_reward_count_adherence_mean": 0.9854166626930236, "eval_reward_count_adherence_std": 0.03405027724802494, "eval_reward_hard_gate_mean": 0.9875, "eval_reward_hard_gate_std": 0.03535533845424652, "eval_reward_repeat_soft_mean": 0.9022143065929413, "eval_reward_repeat_soft_std": 0.07740956265479326, "eval_reward_judge_quality_mean": 0.40862499475479125, "eval_reward_judge_quality_std": 0.15207917941734195, "eval_reward_total_composite_mean": 0.7228510618209839, "eval_reward_total_composite_std": 0.15152148604393006} {"timestamp_utc": "2026-04-13T02:51:34Z", "mode": "train", "global_step": 1801, "epoch": 0.18091411351079859, "loss": 0.053, "grad_norm": 16.234872817993164, "learning_rate": 4.5454545454545455e-06, "num_tokens": 3269596.0, "completions/mean_length": 39.25, "completions/min_length": 34.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.25, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.4882603883743286, "rewards/meter/std": 0.42202138900756836, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9840322732925415, "rewards/repeat_soft/std": 0.01870845817029476, "rewards/judge_quality/mean": 0.5024999976158142, "rewards/judge_quality/std": 0.21224987506866455, "rewards/total_composite/mean": 0.6188703775405884, "rewards/total_composite/std": 0.16125154495239258, "reward": 0.6188703775405884, "reward_std": 0.16125154495239258, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12762057781219482, "sampling/sampling_logp_difference/max": 5.593090534210205, "sampling/importance_sampling_ratio/min": 0.003723502391949296, "sampling/importance_sampling_ratio/mean": 1.0108191967010498, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5597282722592354, "clip_ratio/low_mean": 0.06364589743316174, "clip_ratio/low_min": 0.06364589743316174, "clip_ratio/high_mean": 0.04809158341959119, "clip_ratio/high_max": 0.04809158341959119, "clip_ratio/region_mean": 0.11173748085275292, "reward_total_mean": 0.6188703775405884, "reward_meter_mean": 0.4882603883743286, "reward_meter_std": 0.42202138900756836, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9840322732925415, "reward_repeat_soft_std": 0.01870845817029476, "reward_judge_quality_mean": 0.5024999976158142, "reward_judge_quality_std": 0.21224987506866455, "reward_total_composite_mean": 0.6188703775405884, "reward_total_composite_std": 0.16125154495239258} {"timestamp_utc": "2026-04-13T02:51:46Z", "mode": "train", "global_step": 1802, "epoch": 0.1810145655449523, "loss": -0.0017, "grad_norm": 6.433957576751709, "learning_rate": 4.542424242424243e-06, "num_tokens": 3272126.0, "completions/mean_length": 126.25, "completions/min_length": 117.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.25, "completions/min_terminated_length": 117.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.9037688970565796, "rewards/meter/std": 0.15743732452392578, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8646507263183594, "rewards/repeat_soft/std": 0.07882525026798248, "rewards/judge_quality/mean": 0.26749998331069946, "rewards/judge_quality/std": 0.10375107079744339, "rewards/total_composite/mean": 0.7234110832214355, "rewards/total_composite/std": 0.07757396996021271, "reward": 0.7234110832214355, "reward_std": 0.07757395505905151, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09326555579900742, "sampling/sampling_logp_difference/max": 1.524287223815918, "sampling/importance_sampling_ratio/min": 0.21777622401714325, "sampling/importance_sampling_ratio/mean": 1.002601146697998, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6478887088596821, "clip_ratio/low_mean": 0.021185756660997868, "clip_ratio/low_min": 0.021185756660997868, "clip_ratio/high_mean": 0.08109697513282299, "clip_ratio/high_max": 0.08109697513282299, "clip_ratio/region_mean": 0.10228273179382086, "reward_total_mean": 0.7234110832214355, "reward_meter_mean": 0.9037688970565796, "reward_meter_std": 0.15743732452392578, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8646507263183594, "reward_repeat_soft_std": 0.07882525026798248, "reward_judge_quality_mean": 0.26749998331069946, "reward_judge_quality_std": 0.10375107079744339, "reward_total_composite_mean": 0.7234110832214355, "reward_total_composite_std": 0.07757396996021271} {"timestamp_utc": "2026-04-13T02:51:53Z", "mode": "train", "global_step": 1803, "epoch": 0.18111501757910597, "loss": 0.0979, "grad_norm": 11.374402046203613, "learning_rate": 4.539393939393939e-06, "num_tokens": 3273973.0, "completions/mean_length": 70.875, "completions/min_length": 50.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.875, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.8033242225646973, "rewards/meter/std": 0.19946187734603882, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.17817415297031403, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9360846281051636, "rewards/repeat_soft/std": 0.061100590974092484, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.699729323387146, "rewards/total_composite/std": 0.09262377768754959, "reward": 0.699729323387146, "reward_std": 0.092623770236969, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1330876350402832, "sampling/sampling_logp_difference/max": 1.9872455596923828, "sampling/importance_sampling_ratio/min": 0.13707245886325836, "sampling/importance_sampling_ratio/mean": 0.9940303564071655, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.549153883010149, "clip_ratio/low_mean": 0.04766419529914856, "clip_ratio/low_min": 0.04766419529914856, "clip_ratio/high_mean": 0.0767922019585967, "clip_ratio/high_max": 0.0767922019585967, "clip_ratio/region_mean": 0.12445639725774527, "reward_total_mean": 0.699729323387146, "reward_meter_mean": 0.8033242225646973, "reward_meter_std": 0.19946187734603882, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.17817415297031403, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9360846281051636, "reward_repeat_soft_std": 0.061100590974092484, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.699729323387146, "reward_total_composite_std": 0.09262377768754959} {"timestamp_utc": "2026-04-13T02:52:00Z", "mode": "train", "global_step": 1804, "epoch": 0.18121546961325966, "loss": 0.0177, "grad_norm": 7.621689796447754, "learning_rate": 4.5363636363636364e-06, "num_tokens": 3275746.0, "completions/mean_length": 63.625, "completions/min_length": 57.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.625, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.936316728591919, "rewards/meter/std": 0.13930010795593262, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.923162579536438, "rewards/repeat_soft/std": 0.06680750846862793, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.8271588087081909, "rewards/total_composite/std": 0.10692548751831055, "reward": 0.8271588087081909, "reward_std": 0.10692548006772995, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0916421040892601, "sampling/sampling_logp_difference/max": 1.6845202445983887, "sampling/importance_sampling_ratio/min": 0.18553341925144196, "sampling/importance_sampling_ratio/mean": 0.9921951293945312, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5358054079115391, "clip_ratio/low_mean": 0.07124077808111906, "clip_ratio/low_min": 0.07124077808111906, "clip_ratio/high_mean": 0.045533498749136925, "clip_ratio/high_max": 0.045533498749136925, "clip_ratio/region_mean": 0.11677427683025599, "reward_total_mean": 0.8271588087081909, "reward_meter_mean": 0.936316728591919, "reward_meter_std": 0.13930010795593262, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.923162579536438, "reward_repeat_soft_std": 0.06680750846862793, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.8271588087081909, "reward_total_composite_std": 0.10692548751831055} {"timestamp_utc": "2026-04-13T02:52:15Z", "mode": "train", "global_step": 1805, "epoch": 0.18131592164741336, "loss": -0.0006, "grad_norm": 13.045923233032227, "learning_rate": 4.533333333333334e-06, "num_tokens": 3277507.0, "completions/mean_length": 55.125, "completions/min_length": 49.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.125, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9814106225967407, "rewards/meter/std": 0.009993772953748703, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9354717135429382, "rewards/repeat_soft/std": 0.04626616835594177, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.8048069477081299, "rewards/total_composite/std": 0.02045665681362152, "reward": 0.8048069477081299, "reward_std": 0.020456647500395775, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11236698925495148, "sampling/sampling_logp_difference/max": 1.6863117218017578, "sampling/importance_sampling_ratio/min": 0.18520133197307587, "sampling/importance_sampling_ratio/mean": 1.005398154258728, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6827089488506317, "clip_ratio/low_mean": 0.026535162702202797, "clip_ratio/low_min": 0.026535162702202797, "clip_ratio/high_mean": 0.06190601410344243, "clip_ratio/high_max": 0.06190601410344243, "clip_ratio/region_mean": 0.08844117680564523, "reward_total_mean": 0.8048069477081299, "reward_meter_mean": 0.9814106225967407, "reward_meter_std": 0.009993772953748703, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9354717135429382, "reward_repeat_soft_std": 0.04626616835594177, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.8048069477081299, "reward_total_composite_std": 0.02045665681362152} {"timestamp_utc": "2026-04-13T02:52:27Z", "mode": "train", "global_step": 1806, "epoch": 0.18141637368156704, "loss": 0.0436, "grad_norm": 5.889115333557129, "learning_rate": 4.53030303030303e-06, "num_tokens": 3280597.0, "completions/mean_length": 192.25, "completions/min_length": 156.0, "completions/max_length": 229.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 192.25, "completions/min_terminated_length": 156.0, "completions/max_terminated_length": 229.0, "rewards/meter/mean": 0.40828078985214233, "rewards/meter/std": 0.3778759837150574, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.744086503982544, "rewards/repeat_soft/std": 0.1202196255326271, "rewards/judge_quality/mean": 0.23000000417232513, "rewards/judge_quality/std": 0.12224097549915314, "rewards/total_composite/mean": 0.4708850085735321, "rewards/total_composite/std": 0.1663530170917511, "reward": 0.4708850085735321, "reward_std": 0.1663530170917511, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09750306606292725, "sampling/sampling_logp_difference/max": 1.9115991592407227, "sampling/importance_sampling_ratio/min": 0.14784376323223114, "sampling/importance_sampling_ratio/mean": 1.012341022491455, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.578801691532135, "clip_ratio/low_mean": 0.04552842676639557, "clip_ratio/low_min": 0.04552842676639557, "clip_ratio/high_mean": 0.0386727349832654, "clip_ratio/high_max": 0.0386727349832654, "clip_ratio/region_mean": 0.08420116174966097, "reward_total_mean": 0.4708850085735321, "reward_meter_mean": 0.40828078985214233, "reward_meter_std": 0.3778759837150574, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.744086503982544, "reward_repeat_soft_std": 0.1202196255326271, "reward_judge_quality_mean": 0.23000000417232513, "reward_judge_quality_std": 0.12224097549915314, "reward_total_composite_mean": 0.4708850085735321, "reward_total_composite_std": 0.1663530170917511} {"timestamp_utc": "2026-04-13T02:52:38Z", "mode": "train", "global_step": 1807, "epoch": 0.18151682571572075, "loss": 0.057, "grad_norm": 12.786042213439941, "learning_rate": 4.527272727272727e-06, "num_tokens": 3282005.0, "completions/mean_length": 26.0, "completions/min_length": 19.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.0, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.9676955938339233, "rewards/meter/std": 0.04060090705752373, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8930799961090088, "rewards/repeat_soft/std": 0.16100598871707916, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.8052709698677063, "rewards/total_composite/std": 0.02943754754960537, "reward": 0.8052709698677063, "reward_std": 0.02943754754960537, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11479992419481277, "sampling/sampling_logp_difference/max": 1.3181090354919434, "sampling/importance_sampling_ratio/min": 0.2676409184932709, "sampling/importance_sampling_ratio/mean": 0.9962974190711975, "sampling/importance_sampling_ratio/max": 1.7011643648147583, "entropy": 0.6095785051584244, "clip_ratio/low_mean": 0.022499999962747097, "clip_ratio/low_min": 0.022499999962747097, "clip_ratio/high_mean": 0.09554656082764268, "clip_ratio/high_max": 0.09554656082764268, "clip_ratio/region_mean": 0.11804656079038978, "reward_total_mean": 0.8052709698677063, "reward_meter_mean": 0.9676955938339233, "reward_meter_std": 0.04060090705752373, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8930799961090088, "reward_repeat_soft_std": 0.16100598871707916, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.8052709698677063, "reward_total_composite_std": 0.02943754754960537} {"timestamp_utc": "2026-04-13T02:52:48Z", "mode": "train", "global_step": 1808, "epoch": 0.18161727774987443, "loss": 0.0185, "grad_norm": 7.341400146484375, "learning_rate": 4.524242424242425e-06, "num_tokens": 3284124.0, "completions/mean_length": 97.875, "completions/min_length": 89.0, "completions/max_length": 110.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.875, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 110.0, "rewards/meter/mean": 0.992314338684082, "rewards/meter/std": 0.00552747305482626, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9274189472198486, "rewards/repeat_soft/std": 0.048193417489528656, "rewards/judge_quality/mean": 0.6449999809265137, "rewards/judge_quality/std": 0.24928471446037292, "rewards/total_composite/mean": 0.8827833533287048, "rewards/total_composite/std": 0.0759420171380043, "reward": 0.8827833533287048, "reward_std": 0.0759420171380043, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10079916566610336, "sampling/sampling_logp_difference/max": 1.54884672164917, "sampling/importance_sampling_ratio/min": 0.212492898106575, "sampling/importance_sampling_ratio/mean": 1.007562279701233, "sampling/importance_sampling_ratio/max": 1.9707027673721313, "entropy": 0.5761384963989258, "clip_ratio/low_mean": 0.04480581311509013, "clip_ratio/low_min": 0.04480581311509013, "clip_ratio/high_mean": 0.04814876290038228, "clip_ratio/high_max": 0.04814876290038228, "clip_ratio/region_mean": 0.09295457601547241, "reward_total_mean": 0.8827833533287048, "reward_meter_mean": 0.992314338684082, "reward_meter_std": 0.00552747305482626, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9274189472198486, "reward_repeat_soft_std": 0.048193417489528656, "reward_judge_quality_mean": 0.6449999809265137, "reward_judge_quality_std": 0.24928471446037292, "reward_total_composite_mean": 0.8827833533287048, "reward_total_composite_std": 0.0759420171380043} {"timestamp_utc": "2026-04-13T02:52:55Z", "mode": "train", "global_step": 1809, "epoch": 0.18171772978402811, "loss": 0.0183, "grad_norm": 8.962522506713867, "learning_rate": 4.521212121212122e-06, "num_tokens": 3286018.0, "completions/mean_length": 73.75, "completions/min_length": 68.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.75, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9375483989715576, "rewards/meter/std": 0.11865384131669998, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9575330018997192, "rewards/repeat_soft/std": 0.022606519982218742, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7745250463485718, "rewards/total_composite/std": 0.053457967936992645, "reward": 0.7745250463485718, "reward_std": 0.053457967936992645, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10659663379192352, "sampling/sampling_logp_difference/max": 1.3761329650878906, "sampling/importance_sampling_ratio/min": 0.2525533139705658, "sampling/importance_sampling_ratio/mean": 0.9935368895530701, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5917616933584213, "clip_ratio/low_mean": 0.03783621732145548, "clip_ratio/low_min": 0.03783621732145548, "clip_ratio/high_mean": 0.06953808944672346, "clip_ratio/high_max": 0.06953808944672346, "clip_ratio/region_mean": 0.10737430676817894, "reward_total_mean": 0.7745250463485718, "reward_meter_mean": 0.9375483989715576, "reward_meter_std": 0.11865384131669998, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9575330018997192, "reward_repeat_soft_std": 0.022606519982218742, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7745250463485718, "reward_total_composite_std": 0.053457967936992645} {"timestamp_utc": "2026-04-13T02:53:02Z", "mode": "train", "global_step": 1810, "epoch": 0.18181818181818182, "loss": -0.0222, "grad_norm": 4.4645209312438965, "learning_rate": 4.518181818181819e-06, "num_tokens": 3288694.0, "completions/mean_length": 159.5, "completions/min_length": 142.0, "completions/max_length": 180.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 159.5, "completions/min_terminated_length": 142.0, "completions/max_terminated_length": 180.0, "rewards/meter/mean": 0.8502364754676819, "rewards/meter/std": 0.3303288221359253, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8815233707427979, "rewards/repeat_soft/std": 0.05433381348848343, "rewards/judge_quality/mean": 0.24250000715255737, "rewards/judge_quality/std": 0.11792854964733124, "rewards/total_composite/mean": 0.6897587776184082, "rewards/total_composite/std": 0.12265893071889877, "reward": 0.6897587776184082, "reward_std": 0.12265893071889877, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09413915872573853, "sampling/sampling_logp_difference/max": 1.4247784614562988, "sampling/importance_sampling_ratio/min": 0.24056176841259003, "sampling/importance_sampling_ratio/mean": 1.0049395561218262, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5733625665307045, "clip_ratio/low_mean": 0.018103448674082756, "clip_ratio/low_min": 0.018103448674082756, "clip_ratio/high_mean": 0.08504094649106264, "clip_ratio/high_max": 0.08504094649106264, "clip_ratio/region_mean": 0.1031443951651454, "reward_total_mean": 0.6897587776184082, "reward_meter_mean": 0.8502364754676819, "reward_meter_std": 0.3303288221359253, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8815233707427979, "reward_repeat_soft_std": 0.05433381348848343, "reward_judge_quality_mean": 0.24250000715255737, "reward_judge_quality_std": 0.11792854964733124, "reward_total_composite_mean": 0.6897587776184082, "reward_total_composite_std": 0.12265893071889877} {"timestamp_utc": "2026-04-13T02:53:09Z", "mode": "train", "global_step": 1811, "epoch": 0.1819186338523355, "loss": 0.0536, "grad_norm": 9.042340278625488, "learning_rate": 4.5151515151515155e-06, "num_tokens": 3290581.0, "completions/mean_length": 76.875, "completions/min_length": 65.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.875, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.4035804569721222, "rewards/meter/std": 0.3734830915927887, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.979755163192749, "rewards/repeat_soft/std": 0.01709471270442009, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.1940544992685318, "rewards/total_composite/mean": 0.5690867304801941, "rewards/total_composite/std": 0.14563122391700745, "reward": 0.5690867304801941, "reward_std": 0.14563123881816864, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10597533732652664, "sampling/sampling_logp_difference/max": 1.5291929244995117, "sampling/importance_sampling_ratio/min": 0.21671052277088165, "sampling/importance_sampling_ratio/mean": 1.0068188905715942, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6360190436244011, "clip_ratio/low_mean": 0.07535687601193786, "clip_ratio/low_min": 0.07535687601193786, "clip_ratio/high_mean": 0.02555555570870638, "clip_ratio/high_max": 0.02555555570870638, "clip_ratio/region_mean": 0.10091243172064424, "reward_total_mean": 0.5690867304801941, "reward_meter_mean": 0.4035804569721222, "reward_meter_std": 0.3734830915927887, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.979755163192749, "reward_repeat_soft_std": 0.01709471270442009, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.1940544992685318, "reward_total_composite_mean": 0.5690867304801941, "reward_total_composite_std": 0.14563122391700745} {"timestamp_utc": "2026-04-13T02:53:15Z", "mode": "train", "global_step": 1812, "epoch": 0.1820190858864892, "loss": 0.0076, "grad_norm": 12.747135162353516, "learning_rate": 4.512121212121213e-06, "num_tokens": 3291993.0, "completions/mean_length": 33.5, "completions/min_length": 28.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.5, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.9427045583724976, "rewards/meter/std": 0.12609221041202545, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8895974159240723, "rewards/repeat_soft/std": 0.10815776884555817, "rewards/judge_quality/mean": 0.3137499988079071, "rewards/judge_quality/std": 0.12772038578987122, "rewards/total_composite/mean": 0.7573018074035645, "rewards/total_composite/std": 0.08369968831539154, "reward": 0.7573018074035645, "reward_std": 0.08369969576597214, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14972001314163208, "sampling/sampling_logp_difference/max": 1.7152082920074463, "sampling/importance_sampling_ratio/min": 0.17992624640464783, "sampling/importance_sampling_ratio/mean": 1.0145759582519531, "sampling/importance_sampling_ratio/max": 1.8923479318618774, "entropy": 0.9970297366380692, "clip_ratio/low_mean": 0.06180330226197839, "clip_ratio/low_min": 0.06180330226197839, "clip_ratio/high_mean": 0.07935040444135666, "clip_ratio/high_max": 0.07935040444135666, "clip_ratio/region_mean": 0.14115370670333505, "reward_total_mean": 0.7573018074035645, "reward_meter_mean": 0.9427045583724976, "reward_meter_std": 0.12609221041202545, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8895974159240723, "reward_repeat_soft_std": 0.10815776884555817, "reward_judge_quality_mean": 0.3137499988079071, "reward_judge_quality_std": 0.12772038578987122, "reward_total_composite_mean": 0.7573018074035645, "reward_total_composite_std": 0.08369968831539154} {"timestamp_utc": "2026-04-13T02:53:28Z", "mode": "train", "global_step": 1813, "epoch": 0.1821195379206429, "loss": 0.0462, "grad_norm": 10.869617462158203, "learning_rate": 4.50909090909091e-06, "num_tokens": 3293658.0, "completions/mean_length": 59.125, "completions/min_length": 53.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.125, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.620489776134491, "rewards/meter/std": 0.4378489553928375, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9629998207092285, "rewards/repeat_soft/std": 0.024718835949897766, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.6627703905105591, "rewards/total_composite/std": 0.207564577460289, "reward": 0.6627703905105591, "reward_std": 0.2075645625591278, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11668454110622406, "sampling/sampling_logp_difference/max": 1.4370245933532715, "sampling/importance_sampling_ratio/min": 0.23763376474380493, "sampling/importance_sampling_ratio/mean": 1.0066587924957275, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49221427366137505, "clip_ratio/low_mean": 0.04107159189879894, "clip_ratio/low_min": 0.04107159189879894, "clip_ratio/high_mean": 0.07024973351508379, "clip_ratio/high_max": 0.07024973351508379, "clip_ratio/region_mean": 0.11132132541388273, "reward_total_mean": 0.6627703905105591, "reward_meter_mean": 0.620489776134491, "reward_meter_std": 0.4378489553928375, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9629998207092285, "reward_repeat_soft_std": 0.024718835949897766, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.6627703905105591, "reward_total_composite_std": 0.207564577460289} {"timestamp_utc": "2026-04-13T02:53:39Z", "mode": "train", "global_step": 1814, "epoch": 0.18221998995479657, "loss": 0.0189, "grad_norm": 6.49470329284668, "learning_rate": 4.5060606060606065e-06, "num_tokens": 3295933.0, "completions/mean_length": 106.375, "completions/min_length": 93.0, "completions/max_length": 115.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.375, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 115.0, "rewards/meter/mean": 0.605291485786438, "rewards/meter/std": 0.2786337435245514, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9061664938926697, "rewards/repeat_soft/std": 0.035693731158971786, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.6389977931976318, "rewards/total_composite/std": 0.12454371154308319, "reward": 0.6389977931976318, "reward_std": 0.12454371899366379, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.103157177567482, "sampling/sampling_logp_difference/max": 1.952610969543457, "sampling/importance_sampling_ratio/min": 0.1419030874967575, "sampling/importance_sampling_ratio/mean": 1.0030107498168945, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48530198261141777, "clip_ratio/low_mean": 0.05030977725982666, "clip_ratio/low_min": 0.05030977725982666, "clip_ratio/high_mean": 0.05290443729609251, "clip_ratio/high_max": 0.05290443729609251, "clip_ratio/region_mean": 0.10321421455591917, "reward_total_mean": 0.6389977931976318, "reward_meter_mean": 0.605291485786438, "reward_meter_std": 0.2786337435245514, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9061664938926697, "reward_repeat_soft_std": 0.035693731158971786, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.6389977931976318, "reward_total_composite_std": 0.12454371154308319} {"timestamp_utc": "2026-04-13T02:53:45Z", "mode": "train", "global_step": 1815, "epoch": 0.18232044198895028, "loss": 0.0179, "grad_norm": 7.561637878417969, "learning_rate": 4.503030303030304e-06, "num_tokens": 3297582.0, "completions/mean_length": 51.125, "completions/min_length": 45.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.125, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9576509594917297, "rewards/meter/std": 0.06630495935678482, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9658321142196655, "rewards/repeat_soft/std": 0.04279537498950958, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.8222761154174805, "rewards/total_composite/std": 0.06271738559007645, "reward": 0.8222761154174805, "reward_std": 0.06271739304065704, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09981431812047958, "sampling/sampling_logp_difference/max": 1.2390289306640625, "sampling/importance_sampling_ratio/min": 0.2896653711795807, "sampling/importance_sampling_ratio/mean": 1.0059555768966675, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5239651799201965, "clip_ratio/low_mean": 0.0691394112072885, "clip_ratio/low_min": 0.0691394112072885, "clip_ratio/high_mean": 0.026713515631854534, "clip_ratio/high_max": 0.026713515631854534, "clip_ratio/region_mean": 0.09585292683914304, "reward_total_mean": 0.8222761154174805, "reward_meter_mean": 0.9576509594917297, "reward_meter_std": 0.06630495935678482, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9658321142196655, "reward_repeat_soft_std": 0.04279537498950958, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.8222761154174805, "reward_total_composite_std": 0.06271738559007645} {"timestamp_utc": "2026-04-13T02:53:52Z", "mode": "train", "global_step": 1816, "epoch": 0.18242089402310396, "loss": -0.0113, "grad_norm": 7.1845574378967285, "learning_rate": 4.5e-06, "num_tokens": 3299846.0, "completions/mean_length": 95.0, "completions/min_length": 82.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.0, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.986243724822998, "rewards/meter/std": 0.0045786770060658455, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8695489764213562, "rewards/repeat_soft/std": 0.05235220864415169, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.8116395473480225, "rewards/total_composite/std": 0.03733024746179581, "reward": 0.8116395473480225, "reward_std": 0.0373302586376667, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09310686588287354, "sampling/sampling_logp_difference/max": 3.031846761703491, "sampling/importance_sampling_ratio/min": 0.0482264906167984, "sampling/importance_sampling_ratio/mean": 1.0003105401992798, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45967623218894005, "clip_ratio/low_mean": 0.0582137112505734, "clip_ratio/low_min": 0.0582137112505734, "clip_ratio/high_mean": 0.0273165637627244, "clip_ratio/high_max": 0.0273165637627244, "clip_ratio/region_mean": 0.0855302750132978, "reward_total_mean": 0.8116395473480225, "reward_meter_mean": 0.986243724822998, "reward_meter_std": 0.0045786770060658455, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8695489764213562, "reward_repeat_soft_std": 0.05235220864415169, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.8116395473480225, "reward_total_composite_std": 0.03733024746179581} {"timestamp_utc": "2026-04-13T02:54:03Z", "mode": "train", "global_step": 1817, "epoch": 0.18252134605725767, "loss": -0.1171, "grad_norm": 3.48852801322937, "learning_rate": 4.496969696969697e-06, "num_tokens": 3301726.0, "completions/mean_length": 135.0, "completions/min_length": 74.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 81.14286041259766, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.8617318868637085, "rewards/meter/std": 0.2174583375453949, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9514648914337158, "rewards/repeat_soft/std": 0.037396471947431564, "rewards/judge_quality/mean": 0.3399999737739563, "rewards/judge_quality/std": 0.1505228877067566, "rewards/total_composite/mean": 0.7349258661270142, "rewards/total_composite/std": 0.11621294170618057, "reward": 0.7349258661270142, "reward_std": 0.11621294170618057, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10901206731796265, "sampling/sampling_logp_difference/max": 1.9706785678863525, "sampling/importance_sampling_ratio/min": 0.13936224579811096, "sampling/importance_sampling_ratio/mean": 0.9976006746292114, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4633868634700775, "clip_ratio/low_mean": 0.020866455044597387, "clip_ratio/low_min": 0.020866455044597387, "clip_ratio/high_mean": 0.06662347633391619, "clip_ratio/high_max": 0.06662347633391619, "clip_ratio/region_mean": 0.08748993137851357, "reward_total_mean": 0.7349258661270142, "reward_meter_mean": 0.8617318868637085, "reward_meter_std": 0.2174583375453949, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9514648914337158, "reward_repeat_soft_std": 0.037396471947431564, "reward_judge_quality_mean": 0.3399999737739563, "reward_judge_quality_std": 0.1505228877067566, "reward_total_composite_mean": 0.7349258661270142, "reward_total_composite_std": 0.11621294170618057} {"timestamp_utc": "2026-04-13T02:54:10Z", "mode": "train", "global_step": 1818, "epoch": 0.18262179809141135, "loss": -0.0123, "grad_norm": 14.523164749145508, "learning_rate": 4.493939393939395e-06, "num_tokens": 3303511.0, "completions/mean_length": 52.125, "completions/min_length": 48.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.125, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.8232059478759766, "rewards/meter/std": 0.2647099494934082, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9520491361618042, "rewards/repeat_soft/std": 0.031348563730716705, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7483975887298584, "rewards/total_composite/std": 0.1169702336192131, "reward": 0.7483975887298584, "reward_std": 0.1169702336192131, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09432999044656754, "sampling/sampling_logp_difference/max": 1.4363417625427246, "sampling/importance_sampling_ratio/min": 0.23779608309268951, "sampling/importance_sampling_ratio/mean": 1.003227710723877, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4991745799779892, "clip_ratio/low_mean": 0.042059749364852905, "clip_ratio/low_min": 0.042059749364852905, "clip_ratio/high_mean": 0.08411833271384239, "clip_ratio/high_max": 0.08411833271384239, "clip_ratio/region_mean": 0.1261780820786953, "reward_total_mean": 0.7483975887298584, "reward_meter_mean": 0.8232059478759766, "reward_meter_std": 0.2647099494934082, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9520491361618042, "reward_repeat_soft_std": 0.031348563730716705, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7483975887298584, "reward_total_composite_std": 0.1169702336192131} {"timestamp_utc": "2026-04-13T02:54:17Z", "mode": "train", "global_step": 1819, "epoch": 0.18272225012556503, "loss": 0.0148, "grad_norm": 12.58213996887207, "learning_rate": 4.490909090909091e-06, "num_tokens": 3305204.0, "completions/mean_length": 50.625, "completions/min_length": 48.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.625, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.7526707649230957, "rewards/meter/std": 0.2908581793308258, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9448670148849487, "rewards/repeat_soft/std": 0.046986863017082214, "rewards/judge_quality/mean": 0.48375001549720764, "rewards/judge_quality/std": 0.10809223353862762, "rewards/total_composite/mean": 0.7283135652542114, "rewards/total_composite/std": 0.14433135092258453, "reward": 0.7283135652542114, "reward_std": 0.14433135092258453, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1116800308227539, "sampling/sampling_logp_difference/max": 2.1327450275421143, "sampling/importance_sampling_ratio/min": 0.11851152777671814, "sampling/importance_sampling_ratio/mean": 1.0036882162094116, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5609081834554672, "clip_ratio/low_mean": 0.030833333730697632, "clip_ratio/low_min": 0.030833333730697632, "clip_ratio/high_mean": 0.07935936655849218, "clip_ratio/high_max": 0.07935936655849218, "clip_ratio/region_mean": 0.11019270028918982, "reward_total_mean": 0.7283135652542114, "reward_meter_mean": 0.7526707649230957, "reward_meter_std": 0.2908581793308258, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9448670148849487, "reward_repeat_soft_std": 0.046986863017082214, "reward_judge_quality_mean": 0.48375001549720764, "reward_judge_quality_std": 0.10809223353862762, "reward_total_composite_mean": 0.7283135652542114, "reward_total_composite_std": 0.14433135092258453} {"timestamp_utc": "2026-04-13T02:54:24Z", "mode": "train", "global_step": 1820, "epoch": 0.18282270215971874, "loss": 0.0578, "grad_norm": 9.885247230529785, "learning_rate": 4.487878787878788e-06, "num_tokens": 3306867.0, "completions/mean_length": 51.875, "completions/min_length": 44.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.875, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.6456249952316284, "rewards/meter/std": 0.37667086720466614, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9719966053962708, "rewards/repeat_soft/std": 0.02587537281215191, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.7034809589385986, "rewards/total_composite/std": 0.19792194664478302, "reward": 0.7034809589385986, "reward_std": 0.19792194664478302, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1114538311958313, "sampling/sampling_logp_difference/max": 1.6802372932434082, "sampling/importance_sampling_ratio/min": 0.18632975220680237, "sampling/importance_sampling_ratio/mean": 1.0068477392196655, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5634816624224186, "clip_ratio/low_mean": 0.05488146748393774, "clip_ratio/low_min": 0.05488146748393774, "clip_ratio/high_mean": 0.0837751803919673, "clip_ratio/high_max": 0.0837751803919673, "clip_ratio/region_mean": 0.13865664787590504, "reward_total_mean": 0.7034809589385986, "reward_meter_mean": 0.6456249952316284, "reward_meter_std": 0.37667086720466614, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9719966053962708, "reward_repeat_soft_std": 0.02587537281215191, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.7034809589385986, "reward_total_composite_std": 0.19792194664478302} {"timestamp_utc": "2026-04-13T02:54:30Z", "mode": "train", "global_step": 1821, "epoch": 0.18292315419387242, "loss": 0.0427, "grad_norm": 18.07625961303711, "learning_rate": 4.4848484848484855e-06, "num_tokens": 3308333.0, "completions/mean_length": 35.25, "completions/min_length": 29.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.25, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.7066423892974854, "rewards/meter/std": 0.4114740788936615, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9009312987327576, "rewards/repeat_soft/std": 0.06208154559135437, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.6585822105407715, "rewards/total_composite/std": 0.18598505854606628, "reward": 0.6585822105407715, "reward_std": 0.18598505854606628, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15480409562587738, "sampling/sampling_logp_difference/max": 1.696540355682373, "sampling/importance_sampling_ratio/min": 0.18331663310527802, "sampling/importance_sampling_ratio/mean": 1.0131306648254395, "sampling/importance_sampling_ratio/max": 1.9288514852523804, "entropy": 1.0975374355912209, "clip_ratio/low_mean": 0.02402312681078911, "clip_ratio/low_min": 0.02402312681078911, "clip_ratio/high_mean": 0.11699831136502326, "clip_ratio/high_max": 0.11699831136502326, "clip_ratio/region_mean": 0.14102143817581236, "reward_total_mean": 0.6585822105407715, "reward_meter_mean": 0.7066423892974854, "reward_meter_std": 0.4114740788936615, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9009312987327576, "reward_repeat_soft_std": 0.06208154559135437, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.6585822105407715, "reward_total_composite_std": 0.18598505854606628} {"timestamp_utc": "2026-04-13T02:54:40Z", "mode": "train", "global_step": 1822, "epoch": 0.18302360622802613, "loss": 0.0216, "grad_norm": 9.303369522094727, "learning_rate": 4.481818181818182e-06, "num_tokens": 3310540.0, "completions/mean_length": 109.875, "completions/min_length": 93.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.875, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.7417508959770203, "rewards/meter/std": 0.29277303814888, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.97098308801651, "rewards/repeat_soft/std": 0.013600192032754421, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.6967612504959106, "rewards/total_composite/std": 0.1306823492050171, "reward": 0.6967612504959106, "reward_std": 0.1306823492050171, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13010559976100922, "sampling/sampling_logp_difference/max": 3.4068796634674072, "sampling/importance_sampling_ratio/min": 0.03314446285367012, "sampling/importance_sampling_ratio/mean": 1.016187310218811, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8850555643439293, "clip_ratio/low_mean": 0.04109014570713043, "clip_ratio/low_min": 0.04109014570713043, "clip_ratio/high_mean": 0.07843093946576118, "clip_ratio/high_max": 0.07843093946576118, "clip_ratio/region_mean": 0.11952108517289162, "reward_total_mean": 0.6967612504959106, "reward_meter_mean": 0.7417508959770203, "reward_meter_std": 0.29277303814888, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.97098308801651, "reward_repeat_soft_std": 0.013600192032754421, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.6967612504959106, "reward_total_composite_std": 0.1306823492050171} {"timestamp_utc": "2026-04-13T02:54:47Z", "mode": "train", "global_step": 1823, "epoch": 0.1831240582621798, "loss": 0.029, "grad_norm": 8.59716796875, "learning_rate": 4.478787878787879e-06, "num_tokens": 3312245.0, "completions/mean_length": 61.125, "completions/min_length": 54.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.125, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.6779907941818237, "rewards/meter/std": 0.36607304215431213, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9673361778259277, "rewards/repeat_soft/std": 0.03600212186574936, "rewards/judge_quality/mean": 0.5637500286102295, "rewards/judge_quality/std": 0.22012580931186676, "rewards/total_composite/mean": 0.720954418182373, "rewards/total_composite/std": 0.16165263950824738, "reward": 0.720954418182373, "reward_std": 0.16165263950824738, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11934635788202286, "sampling/sampling_logp_difference/max": 1.701791763305664, "sampling/importance_sampling_ratio/min": 0.18235649168491364, "sampling/importance_sampling_ratio/mean": 1.0108894109725952, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6572780050337315, "clip_ratio/low_mean": 0.03809846006333828, "clip_ratio/low_min": 0.03809846006333828, "clip_ratio/high_mean": 0.0866236463189125, "clip_ratio/high_max": 0.0866236463189125, "clip_ratio/region_mean": 0.12472210638225079, "reward_total_mean": 0.720954418182373, "reward_meter_mean": 0.6779907941818237, "reward_meter_std": 0.36607304215431213, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9673361778259277, "reward_repeat_soft_std": 0.03600212186574936, "reward_judge_quality_mean": 0.5637500286102295, "reward_judge_quality_std": 0.22012580931186676, "reward_total_composite_mean": 0.720954418182373, "reward_total_composite_std": 0.16165263950824738} {"timestamp_utc": "2026-04-13T02:54:53Z", "mode": "train", "global_step": 1824, "epoch": 0.1832245102963335, "loss": 0.0306, "grad_norm": 11.823070526123047, "learning_rate": 4.4757575757575765e-06, "num_tokens": 3313703.0, "completions/mean_length": 32.25, "completions/min_length": 29.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.25, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9872262477874756, "rewards/meter/std": 0.00502823805436492, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9411450624465942, "rewards/repeat_soft/std": 0.030249018222093582, "rewards/judge_quality/mean": 0.3712500035762787, "rewards/judge_quality/std": 0.13695019483566284, "rewards/total_composite/mean": 0.7997413277626038, "rewards/total_composite/std": 0.04210072010755539, "reward": 0.7997413277626038, "reward_std": 0.04210072010755539, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11187545955181122, "sampling/sampling_logp_difference/max": 1.4504222869873047, "sampling/importance_sampling_ratio/min": 0.23447126150131226, "sampling/importance_sampling_ratio/mean": 1.027345895767212, "sampling/importance_sampling_ratio/max": 1.6114871501922607, "entropy": 0.7766570895910263, "clip_ratio/low_mean": 0.038352273404598236, "clip_ratio/low_min": 0.038352273404598236, "clip_ratio/high_mean": 0.07739595975726843, "clip_ratio/high_max": 0.07739595975726843, "clip_ratio/region_mean": 0.11574823316186666, "reward_total_mean": 0.7997413277626038, "reward_meter_mean": 0.9872262477874756, "reward_meter_std": 0.00502823805436492, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9411450624465942, "reward_repeat_soft_std": 0.030249018222093582, "reward_judge_quality_mean": 0.3712500035762787, "reward_judge_quality_std": 0.13695019483566284, "reward_total_composite_mean": 0.7997413277626038, "reward_total_composite_std": 0.04210072010755539} {"timestamp_utc": "2026-04-13T02:54:59Z", "mode": "train", "global_step": 1825, "epoch": 0.1833249623304872, "loss": -0.0006, "grad_norm": 12.547212600708008, "learning_rate": 4.472727272727273e-06, "num_tokens": 3315441.0, "completions/mean_length": 52.25, "completions/min_length": 46.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.25, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.8231343030929565, "rewards/meter/std": 0.2957970201969147, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9330648183822632, "rewards/repeat_soft/std": 0.07876958698034286, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.7595919370651245, "rewards/total_composite/std": 0.1536625325679779, "reward": 0.7595919370651245, "reward_std": 0.1536625325679779, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09930139034986496, "sampling/sampling_logp_difference/max": 1.8741201162338257, "sampling/importance_sampling_ratio/min": 0.15348996222019196, "sampling/importance_sampling_ratio/mean": 1.013654112815857, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5921368673443794, "clip_ratio/low_mean": 0.023993106558918953, "clip_ratio/low_min": 0.023993106558918953, "clip_ratio/high_mean": 0.06852387962862849, "clip_ratio/high_max": 0.06852387962862849, "clip_ratio/region_mean": 0.09251698618754745, "reward_total_mean": 0.7595919370651245, "reward_meter_mean": 0.8231343030929565, "reward_meter_std": 0.2957970201969147, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9330648183822632, "reward_repeat_soft_std": 0.07876958698034286, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.7595919370651245, "reward_total_composite_std": 0.1536625325679779} {"timestamp_utc": "2026-04-13T02:55:06Z", "mode": "train", "global_step": 1826, "epoch": 0.18342541436464088, "loss": 0.0276, "grad_norm": 7.938589572906494, "learning_rate": 4.46969696969697e-06, "num_tokens": 3317496.0, "completions/mean_length": 101.875, "completions/min_length": 93.0, "completions/max_length": 115.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.875, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 115.0, "rewards/meter/mean": 0.9096463322639465, "rewards/meter/std": 0.11657655984163284, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8682117462158203, "rewards/repeat_soft/std": 0.07887614518404007, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7594120502471924, "rewards/total_composite/std": 0.047002099454402924, "reward": 0.7594120502471924, "reward_std": 0.047002099454402924, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09676727652549744, "sampling/sampling_logp_difference/max": 1.4054124355316162, "sampling/importance_sampling_ratio/min": 0.2452658712863922, "sampling/importance_sampling_ratio/mean": 0.994677722454071, "sampling/importance_sampling_ratio/max": 1.7809648513793945, "entropy": 0.5212786011397839, "clip_ratio/low_mean": 0.03649539267644286, "clip_ratio/low_min": 0.03649539267644286, "clip_ratio/high_mean": 0.0514457868412137, "clip_ratio/high_max": 0.0514457868412137, "clip_ratio/region_mean": 0.08794117951765656, "reward_total_mean": 0.7594120502471924, "reward_meter_mean": 0.9096463322639465, "reward_meter_std": 0.11657655984163284, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8682117462158203, "reward_repeat_soft_std": 0.07887614518404007, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7594120502471924, "reward_total_composite_std": 0.047002099454402924} {"timestamp_utc": "2026-04-13T02:55:18Z", "mode": "train", "global_step": 1827, "epoch": 0.18352586639879456, "loss": -0.154, "grad_norm": 2.8647537231445312, "learning_rate": 4.4666666666666665e-06, "num_tokens": 3319313.0, "completions/mean_length": 142.125, "completions/min_length": 80.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 89.28572082519531, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.686698853969574, "rewards/meter/std": 0.38391855359077454, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9751564264297485, "rewards/repeat_soft/std": 0.040571074932813644, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.13274572789669037, "rewards/total_composite/mean": 0.6366551518440247, "rewards/total_composite/std": 0.2834514081478119, "reward": 0.6366551518440247, "reward_std": 0.2834513783454895, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15358206629753113, "sampling/sampling_logp_difference/max": 1.5966720581054688, "sampling/importance_sampling_ratio/min": 0.20256954431533813, "sampling/importance_sampling_ratio/mean": 1.0100536346435547, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6452508121728897, "clip_ratio/low_mean": 0.019801979884505272, "clip_ratio/low_min": 0.019801979884505272, "clip_ratio/high_mean": 0.10977781750261784, "clip_ratio/high_max": 0.10977781750261784, "clip_ratio/region_mean": 0.1295797973871231, "reward_total_mean": 0.6366551518440247, "reward_meter_mean": 0.686698853969574, "reward_meter_std": 0.38391855359077454, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9751564264297485, "reward_repeat_soft_std": 0.040571074932813644, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.13274572789669037, "reward_total_composite_mean": 0.6366551518440247, "reward_total_composite_std": 0.2834514081478119} {"timestamp_utc": "2026-04-13T02:55:25Z", "mode": "train", "global_step": 1828, "epoch": 0.18362631843294827, "loss": 0.0198, "grad_norm": 12.495779037475586, "learning_rate": 4.463636363636364e-06, "num_tokens": 3320988.0, "completions/mean_length": 49.375, "completions/min_length": 42.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.375, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9648280143737793, "rewards/meter/std": 0.03049803525209427, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9591564536094666, "rewards/repeat_soft/std": 0.039455633610486984, "rewards/judge_quality/mean": 0.5225000381469727, "rewards/judge_quality/std": 0.264993280172348, "rewards/total_composite/mean": 0.8368382453918457, "rewards/total_composite/std": 0.07915669679641724, "reward": 0.8368382453918457, "reward_std": 0.07915666699409485, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10936832427978516, "sampling/sampling_logp_difference/max": 1.112809419631958, "sampling/importance_sampling_ratio/min": 0.3286343812942505, "sampling/importance_sampling_ratio/mean": 0.9891215562820435, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6239326559007168, "clip_ratio/low_mean": 0.06857459293678403, "clip_ratio/low_min": 0.06857459293678403, "clip_ratio/high_mean": 0.037253694608807564, "clip_ratio/high_max": 0.037253694608807564, "clip_ratio/region_mean": 0.10582828754559159, "reward_total_mean": 0.8368382453918457, "reward_meter_mean": 0.9648280143737793, "reward_meter_std": 0.03049803525209427, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9591564536094666, "reward_repeat_soft_std": 0.039455633610486984, "reward_judge_quality_mean": 0.5225000381469727, "reward_judge_quality_std": 0.264993280172348, "reward_total_composite_mean": 0.8368382453918457, "reward_total_composite_std": 0.07915669679641724} {"timestamp_utc": "2026-04-13T02:55:32Z", "mode": "train", "global_step": 1829, "epoch": 0.18372677046710195, "loss": 0.0573, "grad_norm": 6.825085163116455, "learning_rate": 4.460606060606061e-06, "num_tokens": 3323228.0, "completions/mean_length": 117.0, "completions/min_length": 106.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.0, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.8580317497253418, "rewards/meter/std": 0.23518714308738708, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9532162547111511, "rewards/repeat_soft/std": 0.03829048573970795, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.20860078930854797, "rewards/total_composite/mean": 0.763435959815979, "rewards/total_composite/std": 0.1229282096028328, "reward": 0.763435959815979, "reward_std": 0.1229282096028328, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11808886379003525, "sampling/sampling_logp_difference/max": 1.7231757640838623, "sampling/importance_sampling_ratio/min": 0.17849837243556976, "sampling/importance_sampling_ratio/mean": 1.011982798576355, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7238490208983421, "clip_ratio/low_mean": 0.04220767319202423, "clip_ratio/low_min": 0.04220767319202423, "clip_ratio/high_mean": 0.07016515266150236, "clip_ratio/high_max": 0.07016515266150236, "clip_ratio/region_mean": 0.11237282585352659, "reward_total_mean": 0.763435959815979, "reward_meter_mean": 0.8580317497253418, "reward_meter_std": 0.23518714308738708, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9532162547111511, "reward_repeat_soft_std": 0.03829048573970795, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.20860078930854797, "reward_total_composite_mean": 0.763435959815979, "reward_total_composite_std": 0.1229282096028328} {"timestamp_utc": "2026-04-13T02:55:39Z", "mode": "train", "global_step": 1830, "epoch": 0.18382722250125566, "loss": 0.0346, "grad_norm": 8.257683753967285, "learning_rate": 4.4575757575757575e-06, "num_tokens": 3325489.0, "completions/mean_length": 111.625, "completions/min_length": 103.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.625, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.8574504852294922, "rewards/meter/std": 0.11071896553039551, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.904655396938324, "rewards/repeat_soft/std": 0.07835724949836731, "rewards/judge_quality/mean": 0.21250000596046448, "rewards/judge_quality/std": 0.0517549142241478, "rewards/total_composite/mean": 0.690068244934082, "rewards/total_composite/std": 0.051468607038259506, "reward": 0.690068244934082, "reward_std": 0.0514686182141304, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10161658376455307, "sampling/sampling_logp_difference/max": 1.880690574645996, "sampling/importance_sampling_ratio/min": 0.15248475968837738, "sampling/importance_sampling_ratio/mean": 1.0147154331207275, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6107008159160614, "clip_ratio/low_mean": 0.05324411392211914, "clip_ratio/low_min": 0.05324411392211914, "clip_ratio/high_mean": 0.03686369676142931, "clip_ratio/high_max": 0.03686369676142931, "clip_ratio/region_mean": 0.09010781068354845, "reward_total_mean": 0.690068244934082, "reward_meter_mean": 0.8574504852294922, "reward_meter_std": 0.11071896553039551, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.904655396938324, "reward_repeat_soft_std": 0.07835724949836731, "reward_judge_quality_mean": 0.21250000596046448, "reward_judge_quality_std": 0.0517549142241478, "reward_total_composite_mean": 0.690068244934082, "reward_total_composite_std": 0.051468607038259506} {"timestamp_utc": "2026-04-13T02:55:46Z", "mode": "train", "global_step": 1831, "epoch": 0.18392767453540934, "loss": -0.0247, "grad_norm": 5.858175754547119, "learning_rate": 4.454545454545455e-06, "num_tokens": 3327630.0, "completions/mean_length": 96.625, "completions/min_length": 86.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.625, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.9863452911376953, "rewards/meter/std": 0.008478543721139431, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9125786423683167, "rewards/repeat_soft/std": 0.053642965853214264, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7983632683753967, "rewards/total_composite/std": 0.028364354744553566, "reward": 0.7983632683753967, "reward_std": 0.028364360332489014, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09385988861322403, "sampling/sampling_logp_difference/max": 2.997058868408203, "sampling/importance_sampling_ratio/min": 0.04993371292948723, "sampling/importance_sampling_ratio/mean": 1.0190401077270508, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5798208378255367, "clip_ratio/low_mean": 0.025839677080512047, "clip_ratio/low_min": 0.025839677080512047, "clip_ratio/high_mean": 0.06099521182477474, "clip_ratio/high_max": 0.06099521182477474, "clip_ratio/region_mean": 0.08683488890528679, "reward_total_mean": 0.7983632683753967, "reward_meter_mean": 0.9863452911376953, "reward_meter_std": 0.008478543721139431, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9125786423683167, "reward_repeat_soft_std": 0.053642965853214264, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7983632683753967, "reward_total_composite_std": 0.028364354744553566} {"timestamp_utc": "2026-04-13T02:55:53Z", "mode": "train", "global_step": 1832, "epoch": 0.18402812656956302, "loss": 0.0203, "grad_norm": 7.415102481842041, "learning_rate": 4.451515151515152e-06, "num_tokens": 3330141.0, "completions/mean_length": 111.875, "completions/min_length": 104.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.875, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.965823769569397, "rewards/meter/std": 0.04492999240756035, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9421778917312622, "rewards/repeat_soft/std": 0.05132538080215454, "rewards/judge_quality/mean": 0.5112500190734863, "rewards/judge_quality/std": 0.268936425447464, "rewards/total_composite/mean": 0.8322135210037231, "rewards/total_composite/std": 0.0878821462392807, "reward": 0.8322135210037231, "reward_std": 0.0878821387887001, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1292385458946228, "sampling/sampling_logp_difference/max": 2.7570042610168457, "sampling/importance_sampling_ratio/min": 0.0634816586971283, "sampling/importance_sampling_ratio/mean": 0.9957097172737122, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6645458340644836, "clip_ratio/low_mean": 0.056127134477719665, "clip_ratio/low_min": 0.056127134477719665, "clip_ratio/high_mean": 0.026118556037545204, "clip_ratio/high_max": 0.026118556037545204, "clip_ratio/region_mean": 0.08224569051526487, "reward_total_mean": 0.8322135210037231, "reward_meter_mean": 0.965823769569397, "reward_meter_std": 0.04492999240756035, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9421778917312622, "reward_repeat_soft_std": 0.05132538080215454, "reward_judge_quality_mean": 0.5112500190734863, "reward_judge_quality_std": 0.268936425447464, "reward_total_composite_mean": 0.8322135210037231, "reward_total_composite_std": 0.0878821462392807} {"timestamp_utc": "2026-04-13T02:56:00Z", "mode": "train", "global_step": 1833, "epoch": 0.18412857860371673, "loss": 0.0326, "grad_norm": 4.864616870880127, "learning_rate": 4.448484848484848e-06, "num_tokens": 3332593.0, "completions/mean_length": 132.5, "completions/min_length": 119.0, "completions/max_length": 146.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.5, "completions/min_terminated_length": 119.0, "completions/max_terminated_length": 146.0, "rewards/meter/mean": 0.975489616394043, "rewards/meter/std": 0.017255980521440506, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8446851968765259, "rewards/repeat_soft/std": 0.060300007462501526, "rewards/judge_quality/mean": 0.22499999403953552, "rewards/judge_quality/std": 0.04629100486636162, "rewards/total_composite/mean": 0.7409389019012451, "rewards/total_composite/std": 0.01655660755932331, "reward": 0.7409389019012451, "reward_std": 0.016556615009903908, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09487338364124298, "sampling/sampling_logp_difference/max": 1.8976030349731445, "sampling/importance_sampling_ratio/min": 0.14992755651474, "sampling/importance_sampling_ratio/mean": 1.0072238445281982, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6483727358281612, "clip_ratio/low_mean": 0.03398909932002425, "clip_ratio/low_min": 0.03398909932002425, "clip_ratio/high_mean": 0.042490565683692694, "clip_ratio/high_max": 0.042490565683692694, "clip_ratio/region_mean": 0.07647966500371695, "reward_total_mean": 0.7409389019012451, "reward_meter_mean": 0.975489616394043, "reward_meter_std": 0.017255980521440506, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8446851968765259, "reward_repeat_soft_std": 0.060300007462501526, "reward_judge_quality_mean": 0.22499999403953552, "reward_judge_quality_std": 0.04629100486636162, "reward_total_composite_mean": 0.7409389019012451, "reward_total_composite_std": 0.01655660755932331} {"timestamp_utc": "2026-04-13T02:56:07Z", "mode": "train", "global_step": 1834, "epoch": 0.1842290306378704, "loss": -0.0101, "grad_norm": 8.798762321472168, "learning_rate": 4.445454545454546e-06, "num_tokens": 3334380.0, "completions/mean_length": 60.375, "completions/min_length": 53.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.375, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9126302003860474, "rewards/meter/std": 0.1354854851961136, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9592065811157227, "rewards/repeat_soft/std": 0.03127916902303696, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7762292623519897, "rewards/total_composite/std": 0.07884057611227036, "reward": 0.7762292623519897, "reward_std": 0.07884056866168976, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1209845244884491, "sampling/sampling_logp_difference/max": 1.1754915714263916, "sampling/importance_sampling_ratio/min": 0.30866721272468567, "sampling/importance_sampling_ratio/mean": 1.0023467540740967, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7782868295907974, "clip_ratio/low_mean": 0.028281912207603455, "clip_ratio/low_min": 0.028281912207603455, "clip_ratio/high_mean": 0.08600028231739998, "clip_ratio/high_max": 0.08600028231739998, "clip_ratio/region_mean": 0.11428219452500343, "reward_total_mean": 0.7762292623519897, "reward_meter_mean": 0.9126302003860474, "reward_meter_std": 0.1354854851961136, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9592065811157227, "reward_repeat_soft_std": 0.03127916902303696, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7762292623519897, "reward_total_composite_std": 0.07884057611227036} {"timestamp_utc": "2026-04-13T02:56:19Z", "mode": "train", "global_step": 1835, "epoch": 0.18432948267202412, "loss": -0.2811, "grad_norm": 2.2742724418640137, "learning_rate": 4.442424242424243e-06, "num_tokens": 3336819.0, "completions/mean_length": 318.875, "completions/min_length": 188.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 203.0, "completions/min_terminated_length": 188.0, "completions/max_terminated_length": 244.0, "rewards/meter/mean": 0.6290279626846313, "rewards/meter/std": 0.26020923256874084, "rewards/count_adherence/mean": 0.6666666269302368, "rewards/count_adherence/std": 0.30860671401023865, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9125152826309204, "rewards/repeat_soft/std": 0.07681315392255783, "rewards/judge_quality/mean": 0.19249999523162842, "rewards/judge_quality/std": 0.1249857097864151, "rewards/total_composite/mean": 0.36234697699546814, "rewards/total_composite/std": 0.302888959646225, "reward": 0.36234697699546814, "reward_std": 0.302888959646225, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09965229779481888, "sampling/sampling_logp_difference/max": 1.929178237915039, "sampling/importance_sampling_ratio/min": 0.1452675312757492, "sampling/importance_sampling_ratio/mean": 1.0101749897003174, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40497900545597076, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06850615050643682, "clip_ratio/high_max": 0.06850615050643682, "clip_ratio/region_mean": 0.06850615050643682, "reward_total_mean": 0.36234697699546814, "reward_meter_mean": 0.6290279626846313, "reward_meter_std": 0.26020923256874084, "reward_count_adherence_mean": 0.6666666269302368, "reward_count_adherence_std": 0.30860671401023865, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9125152826309204, "reward_repeat_soft_std": 0.07681315392255783, "reward_judge_quality_mean": 0.19249999523162842, "reward_judge_quality_std": 0.1249857097864151, "reward_total_composite_mean": 0.36234697699546814, "reward_total_composite_std": 0.302888959646225} {"timestamp_utc": "2026-04-13T02:56:25Z", "mode": "train", "global_step": 1836, "epoch": 0.1844299347061778, "loss": 0.0238, "grad_norm": 8.57689380645752, "learning_rate": 4.43939393939394e-06, "num_tokens": 3338577.0, "completions/mean_length": 57.75, "completions/min_length": 50.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.75, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.8894038796424866, "rewards/meter/std": 0.24728253483772278, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9493073225021362, "rewards/repeat_soft/std": 0.0329010896384716, "rewards/judge_quality/mean": 0.3137499988079071, "rewards/judge_quality/std": 0.12772038578987122, "rewards/total_composite/mean": 0.7392874956130981, "rewards/total_composite/std": 0.13523128628730774, "reward": 0.7392874956130981, "reward_std": 0.13523128628730774, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11549019068479538, "sampling/sampling_logp_difference/max": 1.2181510925292969, "sampling/importance_sampling_ratio/min": 0.2957765460014343, "sampling/importance_sampling_ratio/mean": 1.0245544910430908, "sampling/importance_sampling_ratio/max": 1.8394204378128052, "entropy": 0.8075526282191277, "clip_ratio/low_mean": 0.02517788764089346, "clip_ratio/low_min": 0.02517788764089346, "clip_ratio/high_mean": 0.08762336242944002, "clip_ratio/high_max": 0.08762336242944002, "clip_ratio/region_mean": 0.11280125007033348, "reward_total_mean": 0.7392874956130981, "reward_meter_mean": 0.8894038796424866, "reward_meter_std": 0.24728253483772278, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9493073225021362, "reward_repeat_soft_std": 0.0329010896384716, "reward_judge_quality_mean": 0.3137499988079071, "reward_judge_quality_std": 0.12772038578987122, "reward_total_composite_mean": 0.7392874956130981, "reward_total_composite_std": 0.13523128628730774} {"timestamp_utc": "2026-04-13T02:56:32Z", "mode": "train", "global_step": 1837, "epoch": 0.18453038674033148, "loss": -0.0087, "grad_norm": 6.426721096038818, "learning_rate": 4.436363636363637e-06, "num_tokens": 3341012.0, "completions/mean_length": 112.375, "completions/min_length": 96.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.375, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.912355899810791, "rewards/meter/std": 0.21983936429023743, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7315000295639038, "rewards/repeat_soft/std": 0.057409949600696564, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.7432101964950562, "rewards/total_composite/std": 0.10069253295660019, "reward": 0.7432101964950562, "reward_std": 0.10069253295660019, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07006628811359406, "sampling/sampling_logp_difference/max": 1.9155244827270508, "sampling/importance_sampling_ratio/min": 0.14726456999778748, "sampling/importance_sampling_ratio/mean": 1.003284215927124, "sampling/importance_sampling_ratio/max": 1.9561249017715454, "entropy": 0.36030561476945877, "clip_ratio/low_mean": 0.0065104165114462376, "clip_ratio/low_min": 0.0065104165114462376, "clip_ratio/high_mean": 0.04756954312324524, "clip_ratio/high_max": 0.04756954312324524, "clip_ratio/region_mean": 0.05407995963469148, "reward_total_mean": 0.7432101964950562, "reward_meter_mean": 0.912355899810791, "reward_meter_std": 0.21983936429023743, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7315000295639038, "reward_repeat_soft_std": 0.057409949600696564, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.7432101964950562, "reward_total_composite_std": 0.10069253295660019} {"timestamp_utc": "2026-04-13T02:56:39Z", "mode": "train", "global_step": 1838, "epoch": 0.1846308387744852, "loss": 0.0159, "grad_norm": 9.930435180664062, "learning_rate": 4.433333333333334e-06, "num_tokens": 3342702.0, "completions/mean_length": 48.25, "completions/min_length": 41.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.25, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.9213739633560181, "rewards/meter/std": 0.042237479239702225, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9683213233947754, "rewards/repeat_soft/std": 0.039762724190950394, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.8107004165649414, "rewards/total_composite/std": 0.06022562086582184, "reward": 0.8107004165649414, "reward_std": 0.060225605964660645, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09228319674730301, "sampling/sampling_logp_difference/max": 1.9303498268127441, "sampling/importance_sampling_ratio/min": 0.14509743452072144, "sampling/importance_sampling_ratio/mean": 1.0054072141647339, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48612622171640396, "clip_ratio/low_mean": 0.06461627222597599, "clip_ratio/low_min": 0.06461627222597599, "clip_ratio/high_mean": 0.026041667442768812, "clip_ratio/high_max": 0.026041667442768812, "clip_ratio/region_mean": 0.0906579396687448, "reward_total_mean": 0.8107004165649414, "reward_meter_mean": 0.9213739633560181, "reward_meter_std": 0.042237479239702225, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9683213233947754, "reward_repeat_soft_std": 0.039762724190950394, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.8107004165649414, "reward_total_composite_std": 0.06022562086582184} {"timestamp_utc": "2026-04-13T02:56:45Z", "mode": "train", "global_step": 1839, "epoch": 0.18473129080863887, "loss": 0.0138, "grad_norm": 6.836483001708984, "learning_rate": 4.430303030303031e-06, "num_tokens": 3344417.0, "completions/mean_length": 70.375, "completions/min_length": 59.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.375, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.894707202911377, "rewards/meter/std": 0.2720414400100708, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9121917486190796, "rewards/repeat_soft/std": 0.052724722772836685, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.1249857097864151, "rewards/total_composite/mean": 0.7495874166488647, "rewards/total_composite/std": 0.14726760983467102, "reward": 0.7495874166488647, "reward_std": 0.14726760983467102, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09982019662857056, "sampling/sampling_logp_difference/max": 1.4752569198608398, "sampling/importance_sampling_ratio/min": 0.22871996462345123, "sampling/importance_sampling_ratio/mean": 1.0169264078140259, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7147943899035454, "clip_ratio/low_mean": 0.020397208631038666, "clip_ratio/low_min": 0.020397208631038666, "clip_ratio/high_mean": 0.087316760327667, "clip_ratio/high_max": 0.087316760327667, "clip_ratio/region_mean": 0.10771396895870566, "reward_total_mean": 0.7495874166488647, "reward_meter_mean": 0.894707202911377, "reward_meter_std": 0.2720414400100708, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9121917486190796, "reward_repeat_soft_std": 0.052724722772836685, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.1249857097864151, "reward_total_composite_mean": 0.7495874166488647, "reward_total_composite_std": 0.14726760983467102} {"timestamp_utc": "2026-04-13T02:56:57Z", "mode": "train", "global_step": 1840, "epoch": 0.18483174284279258, "loss": -0.2366, "grad_norm": 1.8936814069747925, "learning_rate": 4.4272727272727275e-06, "num_tokens": 3347491.0, "completions/mean_length": 240.25, "completions/min_length": 189.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 201.42857360839844, "completions/min_terminated_length": 189.0, "completions/max_terminated_length": 217.0, "rewards/meter/mean": 0.8793452978134155, "rewards/meter/std": 0.1950264424085617, "rewards/count_adherence/mean": 0.8500000238418579, "rewards/count_adherence/std": 0.2777460515499115, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8617437481880188, "rewards/repeat_soft/std": 0.11079948395490646, "rewards/judge_quality/mean": 0.16249999403953552, "rewards/judge_quality/std": 0.06408699601888657, "rewards/total_composite/mean": 0.5856037139892578, "rewards/total_composite/std": 0.2537728250026703, "reward": 0.5856037139892578, "reward_std": 0.2537727952003479, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08888793736696243, "sampling/sampling_logp_difference/max": 1.2528724670410156, "sampling/importance_sampling_ratio/min": 0.2856829762458801, "sampling/importance_sampling_ratio/mean": 1.0176150798797607, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6961172521114349, "clip_ratio/low_mean": 0.008373205550014973, "clip_ratio/low_min": 0.008373205550014973, "clip_ratio/high_mean": 0.06987863313406706, "clip_ratio/high_max": 0.06987863313406706, "clip_ratio/region_mean": 0.07825183868408203, "reward_total_mean": 0.5856037139892578, "reward_meter_mean": 0.8793452978134155, "reward_meter_std": 0.1950264424085617, "reward_count_adherence_mean": 0.8500000238418579, "reward_count_adherence_std": 0.2777460515499115, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8617437481880188, "reward_repeat_soft_std": 0.11079948395490646, "reward_judge_quality_mean": 0.16249999403953552, "reward_judge_quality_std": 0.06408699601888657, "reward_total_composite_mean": 0.5856037139892578, "reward_total_composite_std": 0.2537728250026703} {"timestamp_utc": "2026-04-13T02:57:04Z", "mode": "train", "global_step": 1841, "epoch": 0.18493219487694626, "loss": 0.0055, "grad_norm": 9.057662010192871, "learning_rate": 4.424242424242425e-06, "num_tokens": 3349063.0, "completions/mean_length": 47.5, "completions/min_length": 42.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.5, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.893686056137085, "rewards/meter/std": 0.19734446704387665, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9868874549865723, "rewards/repeat_soft/std": 0.009955487214028835, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.780222475528717, "rewards/total_composite/std": 0.08970384299755096, "reward": 0.780222475528717, "reward_std": 0.08970382809638977, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13147002458572388, "sampling/sampling_logp_difference/max": 1.4674577713012695, "sampling/importance_sampling_ratio/min": 0.23051075637340546, "sampling/importance_sampling_ratio/mean": 1.0174201726913452, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7685233354568481, "clip_ratio/low_mean": 0.05029946751892567, "clip_ratio/low_min": 0.05029946751892567, "clip_ratio/high_mean": 0.10382651723921299, "clip_ratio/high_max": 0.10382651723921299, "clip_ratio/region_mean": 0.15412598475813866, "reward_total_mean": 0.780222475528717, "reward_meter_mean": 0.893686056137085, "reward_meter_std": 0.19734446704387665, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9868874549865723, "reward_repeat_soft_std": 0.009955487214028835, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.780222475528717, "reward_total_composite_std": 0.08970384299755096} {"timestamp_utc": "2026-04-13T02:57:12Z", "mode": "train", "global_step": 1842, "epoch": 0.18503264691109994, "loss": -0.0104, "grad_norm": 4.039283275604248, "learning_rate": 4.421212121212122e-06, "num_tokens": 3351869.0, "completions/mean_length": 176.75, "completions/min_length": 168.0, "completions/max_length": 196.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 176.75, "completions/min_terminated_length": 168.0, "completions/max_terminated_length": 196.0, "rewards/meter/mean": 0.9826877117156982, "rewards/meter/std": 0.014017526991665363, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8105181455612183, "rewards/repeat_soft/std": 0.08791357278823853, "rewards/judge_quality/mean": 0.2212499976158142, "rewards/judge_quality/std": 0.09433034062385559, "rewards/total_composite/mean": 0.7396363019943237, "rewards/total_composite/std": 0.031248530372977257, "reward": 0.7396363019943237, "reward_std": 0.031248528510332108, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08458785712718964, "sampling/sampling_logp_difference/max": 1.8345746994018555, "sampling/importance_sampling_ratio/min": 0.15968140959739685, "sampling/importance_sampling_ratio/mean": 1.0096619129180908, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6123048327863216, "clip_ratio/low_mean": 0.030781902838498354, "clip_ratio/low_min": 0.030781902838498354, "clip_ratio/high_mean": 0.037117536179721355, "clip_ratio/high_max": 0.037117536179721355, "clip_ratio/region_mean": 0.06789943901821971, "reward_total_mean": 0.7396363019943237, "reward_meter_mean": 0.9826877117156982, "reward_meter_std": 0.014017526991665363, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8105181455612183, "reward_repeat_soft_std": 0.08791357278823853, "reward_judge_quality_mean": 0.2212499976158142, "reward_judge_quality_std": 0.09433034062385559, "reward_total_composite_mean": 0.7396363019943237, "reward_total_composite_std": 0.031248530372977257} {"timestamp_utc": "2026-04-13T02:57:18Z", "mode": "train", "global_step": 1843, "epoch": 0.18513309894525365, "loss": 0.0426, "grad_norm": 13.98814868927002, "learning_rate": 4.418181818181818e-06, "num_tokens": 3353621.0, "completions/mean_length": 43.0, "completions/min_length": 33.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9200463891029358, "rewards/meter/std": 0.09440668672323227, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8872338533401489, "rewards/repeat_soft/std": 0.17432603240013123, "rewards/judge_quality/mean": 0.38874998688697815, "rewards/judge_quality/std": 0.08675704896450043, "rewards/total_composite/mean": 0.7693692445755005, "rewards/total_composite/std": 0.04314259812235832, "reward": 0.7693692445755005, "reward_std": 0.04314257577061653, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0881357342004776, "sampling/sampling_logp_difference/max": 1.5298689603805542, "sampling/importance_sampling_ratio/min": 0.21656405925750732, "sampling/importance_sampling_ratio/mean": 0.9930169582366943, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4168389029800892, "clip_ratio/low_mean": 0.02198067680001259, "clip_ratio/low_min": 0.02198067680001259, "clip_ratio/high_mean": 0.05424442049115896, "clip_ratio/high_max": 0.05424442049115896, "clip_ratio/region_mean": 0.07622509729117155, "reward_total_mean": 0.7693692445755005, "reward_meter_mean": 0.9200463891029358, "reward_meter_std": 0.09440668672323227, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8872338533401489, "reward_repeat_soft_std": 0.17432603240013123, "reward_judge_quality_mean": 0.38874998688697815, "reward_judge_quality_std": 0.08675704896450043, "reward_total_composite_mean": 0.7693692445755005, "reward_total_composite_std": 0.04314259812235832} {"timestamp_utc": "2026-04-13T02:57:25Z", "mode": "train", "global_step": 1844, "epoch": 0.18523355097940733, "loss": 0.0848, "grad_norm": 6.078597068786621, "learning_rate": 4.415151515151516e-06, "num_tokens": 3356323.0, "completions/mean_length": 145.75, "completions/min_length": 128.0, "completions/max_length": 163.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 145.75, "completions/min_terminated_length": 128.0, "completions/max_terminated_length": 163.0, "rewards/meter/mean": 0.7608742713928223, "rewards/meter/std": 0.2467881441116333, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7853513956069946, "rewards/repeat_soft/std": 0.08711767196655273, "rewards/judge_quality/mean": 0.2887499928474426, "rewards/judge_quality/std": 0.11630470305681229, "rewards/total_composite/mean": 0.6481785774230957, "rewards/total_composite/std": 0.141428604722023, "reward": 0.6481785774230957, "reward_std": 0.141428604722023, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09329482167959213, "sampling/sampling_logp_difference/max": 1.1874194145202637, "sampling/importance_sampling_ratio/min": 0.30500736832618713, "sampling/importance_sampling_ratio/mean": 1.001253604888916, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6455314792692661, "clip_ratio/low_mean": 0.02550099929794669, "clip_ratio/low_min": 0.02550099929794669, "clip_ratio/high_mean": 0.05783093348145485, "clip_ratio/high_max": 0.05783093348145485, "clip_ratio/region_mean": 0.08333193277940154, "reward_total_mean": 0.6481785774230957, "reward_meter_mean": 0.7608742713928223, "reward_meter_std": 0.2467881441116333, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7853513956069946, "reward_repeat_soft_std": 0.08711767196655273, "reward_judge_quality_mean": 0.2887499928474426, "reward_judge_quality_std": 0.11630470305681229, "reward_total_composite_mean": 0.6481785774230957, "reward_total_composite_std": 0.141428604722023} {"timestamp_utc": "2026-04-13T02:57:37Z", "mode": "train", "global_step": 1845, "epoch": 0.18533400301356104, "loss": -0.1401, "grad_norm": 2.2413418292999268, "learning_rate": 4.412121212121213e-06, "num_tokens": 3357966.0, "completions/mean_length": 111.375, "completions/min_length": 49.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 54.142860412597656, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9605326652526855, "rewards/meter/std": 0.07407969981431961, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9586344957351685, "rewards/repeat_soft/std": 0.02958778291940689, "rewards/judge_quality/mean": 0.42500001192092896, "rewards/judge_quality/std": 0.3426785469055176, "rewards/total_composite/mean": 0.7310754060745239, "rewards/total_composite/std": 0.30908554792404175, "reward": 0.7310754060745239, "reward_std": 0.30908554792404175, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13570956885814667, "sampling/sampling_logp_difference/max": 1.3658838272094727, "sampling/importance_sampling_ratio/min": 0.2551550567150116, "sampling/importance_sampling_ratio/mean": 1.026092290878296, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8869519084692001, "clip_ratio/low_mean": 0.009259259328246117, "clip_ratio/low_min": 0.009259259328246117, "clip_ratio/high_mean": 0.09549316996708512, "clip_ratio/high_max": 0.09549316996708512, "clip_ratio/region_mean": 0.10475242929533124, "reward_total_mean": 0.7310754060745239, "reward_meter_mean": 0.9605326652526855, "reward_meter_std": 0.07407969981431961, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9586344957351685, "reward_repeat_soft_std": 0.02958778291940689, "reward_judge_quality_mean": 0.42500001192092896, "reward_judge_quality_std": 0.3426785469055176, "reward_total_composite_mean": 0.7310754060745239, "reward_total_composite_std": 0.30908554792404175} {"timestamp_utc": "2026-04-13T02:57:45Z", "mode": "train", "global_step": 1846, "epoch": 0.18543445504771472, "loss": 0.0696, "grad_norm": 6.240595817565918, "learning_rate": 4.409090909090909e-06, "num_tokens": 3360690.0, "completions/mean_length": 140.5, "completions/min_length": 131.0, "completions/max_length": 159.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 140.5, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 159.0, "rewards/meter/mean": 0.6482356786727905, "rewards/meter/std": 0.36325401067733765, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8134457468986511, "rewards/repeat_soft/std": 0.07109523564577103, "rewards/judge_quality/mean": 0.2762500047683716, "rewards/judge_quality/std": 0.12603145837783813, "rewards/total_composite/mean": 0.6012381315231323, "rewards/total_composite/std": 0.1834052950143814, "reward": 0.6012381315231323, "reward_std": 0.1834052950143814, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07736997306346893, "sampling/sampling_logp_difference/max": 1.6778061389923096, "sampling/importance_sampling_ratio/min": 0.1867832988500595, "sampling/importance_sampling_ratio/mean": 1.0063228607177734, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4604159854352474, "clip_ratio/low_mean": 0.02556779608130455, "clip_ratio/low_min": 0.02556779608130455, "clip_ratio/high_mean": 0.04936626087874174, "clip_ratio/high_max": 0.04936626087874174, "clip_ratio/region_mean": 0.07493405696004629, "reward_total_mean": 0.6012381315231323, "reward_meter_mean": 0.6482356786727905, "reward_meter_std": 0.36325401067733765, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8134457468986511, "reward_repeat_soft_std": 0.07109523564577103, "reward_judge_quality_mean": 0.2762500047683716, "reward_judge_quality_std": 0.12603145837783813, "reward_total_composite_mean": 0.6012381315231323, "reward_total_composite_std": 0.1834052950143814} {"timestamp_utc": "2026-04-13T02:57:52Z", "mode": "train", "global_step": 1847, "epoch": 0.1855349070818684, "loss": -0.0664, "grad_norm": 10.328042030334473, "learning_rate": 4.4060606060606066e-06, "num_tokens": 3362452.0, "completions/mean_length": 59.25, "completions/min_length": 50.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.25, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.8035181760787964, "rewards/meter/std": 0.2931819260120392, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9770846962928772, "rewards/repeat_soft/std": 0.02023920975625515, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.0843462198972702, "rewards/total_composite/mean": 0.7247916460037231, "rewards/total_composite/std": 0.12630081176757812, "reward": 0.7247916460037231, "reward_std": 0.12630079686641693, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11615701019763947, "sampling/sampling_logp_difference/max": 2.045048713684082, "sampling/importance_sampling_ratio/min": 0.12937387824058533, "sampling/importance_sampling_ratio/mean": 0.9771414995193481, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5847748331725597, "clip_ratio/low_mean": 0.039984818547964096, "clip_ratio/low_min": 0.039984818547964096, "clip_ratio/high_mean": 0.08600639551877975, "clip_ratio/high_max": 0.08600639551877975, "clip_ratio/region_mean": 0.12599121406674385, "reward_total_mean": 0.7247916460037231, "reward_meter_mean": 0.8035181760787964, "reward_meter_std": 0.2931819260120392, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9770846962928772, "reward_repeat_soft_std": 0.02023920975625515, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.0843462198972702, "reward_total_composite_mean": 0.7247916460037231, "reward_total_composite_std": 0.12630081176757812} {"timestamp_utc": "2026-04-13T02:58:03Z", "mode": "train", "global_step": 1848, "epoch": 0.1856353591160221, "loss": -0.0422, "grad_norm": 3.875032424926758, "learning_rate": 4.403030303030304e-06, "num_tokens": 3363856.0, "completions/mean_length": 86.5, "completions/min_length": 22.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 25.71428680419922, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.6578601598739624, "rewards/meter/std": 0.45493215322494507, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9606159925460815, "rewards/repeat_soft/std": 0.0053286682814359665, "rewards/judge_quality/mean": 0.24250000715255737, "rewards/judge_quality/std": 0.12947696447372437, "rewards/total_composite/mean": 0.6148486733436584, "rewards/total_composite/std": 0.23073874413967133, "reward": 0.6148486733436584, "reward_std": 0.23073874413967133, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15053842961788177, "sampling/sampling_logp_difference/max": 1.0427300930023193, "sampling/importance_sampling_ratio/min": 0.35249102115631104, "sampling/importance_sampling_ratio/mean": 1.0182132720947266, "sampling/importance_sampling_ratio/max": 1.8157347440719604, "entropy": 0.8333540111780167, "clip_ratio/low_mean": 0.023682336322963238, "clip_ratio/low_min": 0.023682336322963238, "clip_ratio/high_mean": 0.07791456207633018, "clip_ratio/high_max": 0.07791456207633018, "clip_ratio/region_mean": 0.10159689839929342, "reward_total_mean": 0.6148486733436584, "reward_meter_mean": 0.6578601598739624, "reward_meter_std": 0.45493215322494507, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9606159925460815, "reward_repeat_soft_std": 0.0053286682814359665, "reward_judge_quality_mean": 0.24250000715255737, "reward_judge_quality_std": 0.12947696447372437, "reward_total_composite_mean": 0.6148486733436584, "reward_total_composite_std": 0.23073874413967133} {"timestamp_utc": "2026-04-13T02:58:09Z", "mode": "train", "global_step": 1849, "epoch": 0.1857358111501758, "loss": 0.0209, "grad_norm": 6.4094977378845215, "learning_rate": 4.4e-06, "num_tokens": 3365649.0, "completions/mean_length": 62.125, "completions/min_length": 58.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.125, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9836244583129883, "rewards/meter/std": 0.012823912315070629, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7269096970558167, "rewards/repeat_soft/std": 0.14649903774261475, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.19334925711154938, "rewards/total_composite/mean": 0.805946946144104, "rewards/total_composite/std": 0.057003527879714966, "reward": 0.805946946144104, "reward_std": 0.05700352415442467, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10477995872497559, "sampling/sampling_logp_difference/max": 1.9388689994812012, "sampling/importance_sampling_ratio/min": 0.14386656880378723, "sampling/importance_sampling_ratio/mean": 1.0154012441635132, "sampling/importance_sampling_ratio/max": 1.9291588068008423, "entropy": 0.653239332139492, "clip_ratio/low_mean": 0.06806555995717645, "clip_ratio/low_min": 0.06806555995717645, "clip_ratio/high_mean": 0.030374139547348022, "clip_ratio/high_max": 0.030374139547348022, "clip_ratio/region_mean": 0.09843969950452447, "reward_total_mean": 0.805946946144104, "reward_meter_mean": 0.9836244583129883, "reward_meter_std": 0.012823912315070629, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7269096970558167, "reward_repeat_soft_std": 0.14649903774261475, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.19334925711154938, "reward_total_composite_mean": 0.805946946144104, "reward_total_composite_std": 0.057003527879714966} {"timestamp_utc": "2026-04-13T02:58:21Z", "mode": "train", "global_step": 1850, "epoch": 0.18583626318432947, "loss": -0.208, "grad_norm": 1.7132749557495117, "learning_rate": 4.3969696969696975e-06, "num_tokens": 3367775.0, "completions/mean_length": 169.75, "completions/min_length": 101.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 120.85714721679688, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.9854386448860168, "rewards/meter/std": 0.01446529570966959, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8164437413215637, "rewards/repeat_soft/std": 0.12056206911802292, "rewards/judge_quality/mean": 0.23874999582767487, "rewards/judge_quality/std": 0.15384940803050995, "rewards/total_composite/mean": 0.6575117111206055, "rewards/total_composite/std": 0.2691360116004944, "reward": 0.6575117111206055, "reward_std": 0.2691360116004944, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07571952044963837, "sampling/sampling_logp_difference/max": 2.1372880935668945, "sampling/importance_sampling_ratio/min": 0.11797434836626053, "sampling/importance_sampling_ratio/mean": 1.0081895589828491, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.338854618370533, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.07293788017705083, "clip_ratio/high_max": 0.07293788017705083, "clip_ratio/region_mean": 0.07293788017705083, "reward_total_mean": 0.6575117111206055, "reward_meter_mean": 0.9854386448860168, "reward_meter_std": 0.01446529570966959, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.2651650309562683, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8164437413215637, "reward_repeat_soft_std": 0.12056206911802292, "reward_judge_quality_mean": 0.23874999582767487, "reward_judge_quality_std": 0.15384940803050995, "reward_total_composite_mean": 0.6575117111206055, "reward_total_composite_std": 0.2691360116004944} {"timestamp_utc": "2026-04-13T02:59:23Z", "mode": "eval", "global_step": 1850, "epoch": 0.18583626318432947, "eval_loss": NaN, "eval_runtime": 61.1751, "eval_samples_per_second": 1.308, "eval_steps_per_second": 0.163, "eval_num_tokens": 3367775.0, "eval_completions/mean_length": 115.8375, "eval_completions/min_length": 43.2, "eval_completions/max_length": 252.8, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 106.02142944335938, "eval_completions/min_terminated_length": 43.2, "eval_completions/max_terminated_length": 194.4, "eval_rewards/meter/mean": 0.8162377178668976, "eval_rewards/meter/std": 0.28363151401281356, "eval_rewards/count_adherence/mean": 0.9737499833106995, "eval_rewards/count_adherence/std": 0.053273120522499086, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.07071067690849304, "eval_rewards/repeat_soft/mean": 0.8767021656036377, "eval_rewards/repeat_soft/std": 0.1035081660374999, "eval_rewards/judge_quality/mean": 0.29749999940395355, "eval_rewards/judge_quality/std": 0.12763721868395805, "eval_rewards/total_composite/mean": 0.6789032042026519, "eval_rewards/total_composite/std": 0.1643088199198246, "eval_reward": 0.6789032042026519, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.05728467367589474, "eval_sampling/sampling_logp_difference/max": 0.952991247177124, "eval_sampling/importance_sampling_ratio/min": 0.394000244140625, "eval_sampling/importance_sampling_ratio/mean": 1.0091064691543579, "eval_sampling/importance_sampling_ratio/max": 1.3765904307365417, "eval_entropy": 0.5994600921869278, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6789032042026519, "eval_reward_meter_mean": 0.8162377178668976, "eval_reward_meter_std": 0.28363151401281356, "eval_reward_count_adherence_mean": 0.9737499833106995, "eval_reward_count_adherence_std": 0.053273120522499086, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.07071067690849304, "eval_reward_repeat_soft_mean": 0.8767021656036377, "eval_reward_repeat_soft_std": 0.1035081660374999, "eval_reward_judge_quality_mean": 0.29749999940395355, "eval_reward_judge_quality_std": 0.12763721868395805, "eval_reward_total_composite_mean": 0.6789032042026519, "eval_reward_total_composite_std": 0.1643088199198246} {"timestamp_utc": "2026-04-13T02:59:33Z", "mode": "train", "global_step": 1851, "epoch": 0.18593671521848318, "loss": 0.094, "grad_norm": 8.490787506103516, "learning_rate": 4.393939393939394e-06, "num_tokens": 3369795.0, "completions/mean_length": 86.5, "completions/min_length": 76.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.5, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.6573423743247986, "rewards/meter/std": 0.3627721965312958, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8438625335693359, "rewards/repeat_soft/std": 0.09276197105646133, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.12351980805397034, "rewards/total_composite/mean": 0.6231902837753296, "rewards/total_composite/std": 0.17540216445922852, "reward": 0.6231902837753296, "reward_std": 0.17540216445922852, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1123739555478096, "sampling/sampling_logp_difference/max": 2.098461627960205, "sampling/importance_sampling_ratio/min": 0.12264495342969894, "sampling/importance_sampling_ratio/mean": 1.011771321296692, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5617088675498962, "clip_ratio/low_mean": 0.03496686276048422, "clip_ratio/low_min": 0.03496686276048422, "clip_ratio/high_mean": 0.06432341551408172, "clip_ratio/high_max": 0.06432341551408172, "clip_ratio/region_mean": 0.09929027827456594, "reward_total_mean": 0.6231902837753296, "reward_meter_mean": 0.6573423743247986, "reward_meter_std": 0.3627721965312958, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8438625335693359, "reward_repeat_soft_std": 0.09276197105646133, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.12351980805397034, "reward_total_composite_mean": 0.6231902837753296, "reward_total_composite_std": 0.17540216445922852} {"timestamp_utc": "2026-04-13T02:59:45Z", "mode": "train", "global_step": 1852, "epoch": 0.18603716725263686, "loss": -0.1666, "grad_norm": 2.1983132362365723, "learning_rate": 4.390909090909091e-06, "num_tokens": 3371575.0, "completions/mean_length": 123.5, "completions/min_length": 59.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.8937624096870422, "rewards/meter/std": 0.19415977597236633, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9384405612945557, "rewards/repeat_soft/std": 0.03662831336259842, "rewards/judge_quality/mean": 0.24249999225139618, "rewards/judge_quality/std": 0.12947696447372437, "rewards/total_composite/mean": 0.6622787714004517, "rewards/total_composite/std": 0.2702905535697937, "reward": 0.6622787714004517, "reward_std": 0.2702905535697937, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12411025166511536, "sampling/sampling_logp_difference/max": 1.7102290391921997, "sampling/importance_sampling_ratio/min": 0.1808243691921234, "sampling/importance_sampling_ratio/mean": 1.0134193897247314, "sampling/importance_sampling_ratio/max": 1.9051697254180908, "entropy": 0.7092911079525948, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10810152627527714, "clip_ratio/high_max": 0.10810152627527714, "clip_ratio/region_mean": 0.10810152627527714, "reward_total_mean": 0.6622787714004517, "reward_meter_mean": 0.8937624096870422, "reward_meter_std": 0.19415977597236633, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9384405612945557, "reward_repeat_soft_std": 0.03662831336259842, "reward_judge_quality_mean": 0.24249999225139618, "reward_judge_quality_std": 0.12947696447372437, "reward_total_composite_mean": 0.6622787714004517, "reward_total_composite_std": 0.2702905535697937} {"timestamp_utc": "2026-04-13T02:59:54Z", "mode": "train", "global_step": 1853, "epoch": 0.18613761928679057, "loss": 0.0178, "grad_norm": 5.049228668212891, "learning_rate": 4.387878787878788e-06, "num_tokens": 3374892.0, "completions/mean_length": 225.625, "completions/min_length": 208.0, "completions/max_length": 263.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 225.625, "completions/min_terminated_length": 208.0, "completions/max_terminated_length": 263.0, "rewards/meter/mean": 0.9897759556770325, "rewards/meter/std": 0.007201417814940214, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6609230041503906, "rewards/repeat_soft/std": 0.21972531080245972, "rewards/judge_quality/mean": 0.26375001668930054, "rewards/judge_quality/std": 0.13373079895973206, "rewards/total_composite/mean": 0.7374914884567261, "rewards/total_composite/std": 0.061487119644880295, "reward": 0.7374914884567261, "reward_std": 0.061487115919589996, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05882082134485245, "sampling/sampling_logp_difference/max": 2.143890380859375, "sampling/importance_sampling_ratio/min": 0.11719801276922226, "sampling/importance_sampling_ratio/mean": 1.0059717893600464, "sampling/importance_sampling_ratio/max": 1.8108559846878052, "entropy": 0.4085807744413614, "clip_ratio/low_mean": 0.01698610046878457, "clip_ratio/low_min": 0.01698610046878457, "clip_ratio/high_mean": 0.03126882156357169, "clip_ratio/high_max": 0.03126882156357169, "clip_ratio/region_mean": 0.04825492203235626, "reward_total_mean": 0.7374914884567261, "reward_meter_mean": 0.9897759556770325, "reward_meter_std": 0.007201417814940214, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6609230041503906, "reward_repeat_soft_std": 0.21972531080245972, "reward_judge_quality_mean": 0.26375001668930054, "reward_judge_quality_std": 0.13373079895973206, "reward_total_composite_mean": 0.7374914884567261, "reward_total_composite_std": 0.061487119644880295} {"timestamp_utc": "2026-04-13T03:00:01Z", "mode": "train", "global_step": 1854, "epoch": 0.18623807132094425, "loss": 0.0259, "grad_norm": 6.621765613555908, "learning_rate": 4.384848484848485e-06, "num_tokens": 3377023.0, "completions/mean_length": 100.375, "completions/min_length": 95.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.375, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.451884925365448, "rewards/meter/std": 0.279491126537323, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8053433299064636, "rewards/repeat_soft/std": 0.16555792093276978, "rewards/judge_quality/mean": 0.3812500238418579, "rewards/judge_quality/std": 0.25542333722114563, "rewards/total_composite/mean": 0.5482575297355652, "rewards/total_composite/std": 0.197259321808815, "reward": 0.5482575297355652, "reward_std": 0.1972593367099762, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10244850069284439, "sampling/sampling_logp_difference/max": 1.778921127319336, "sampling/importance_sampling_ratio/min": 0.16882018744945526, "sampling/importance_sampling_ratio/mean": 1.0072004795074463, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5857124365866184, "clip_ratio/low_mean": 0.05049439147114754, "clip_ratio/low_min": 0.05049439147114754, "clip_ratio/high_mean": 0.03557831048965454, "clip_ratio/high_max": 0.03557831048965454, "clip_ratio/region_mean": 0.08607270196080208, "reward_total_mean": 0.5482575297355652, "reward_meter_mean": 0.451884925365448, "reward_meter_std": 0.279491126537323, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8053433299064636, "reward_repeat_soft_std": 0.16555792093276978, "reward_judge_quality_mean": 0.3812500238418579, "reward_judge_quality_std": 0.25542333722114563, "reward_total_composite_mean": 0.5482575297355652, "reward_total_composite_std": 0.197259321808815} {"timestamp_utc": "2026-04-13T03:00:08Z", "mode": "train", "global_step": 1855, "epoch": 0.18633852335509793, "loss": 0.0445, "grad_norm": 12.130240440368652, "learning_rate": 4.381818181818182e-06, "num_tokens": 3378859.0, "completions/mean_length": 66.5, "completions/min_length": 60.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.6585661172866821, "rewards/meter/std": 0.40065932273864746, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9159197807312012, "rewards/repeat_soft/std": 0.06366945058107376, "rewards/judge_quality/mean": 0.4012500047683716, "rewards/judge_quality/std": 0.22793717682361603, "rewards/total_composite/mean": 0.658321738243103, "rewards/total_composite/std": 0.14894212782382965, "reward": 0.658321738243103, "reward_std": 0.14894212782382965, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10254836082458496, "sampling/sampling_logp_difference/max": 4.344371795654297, "sampling/importance_sampling_ratio/min": 0.0129796601831913, "sampling/importance_sampling_ratio/mean": 1.002854585647583, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5000428967177868, "clip_ratio/low_mean": 0.04776283842511475, "clip_ratio/low_min": 0.04776283842511475, "clip_ratio/high_mean": 0.0500065591186285, "clip_ratio/high_max": 0.0500065591186285, "clip_ratio/region_mean": 0.09776939754374325, "reward_total_mean": 0.658321738243103, "reward_meter_mean": 0.6585661172866821, "reward_meter_std": 0.40065932273864746, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9159197807312012, "reward_repeat_soft_std": 0.06366945058107376, "reward_judge_quality_mean": 0.4012500047683716, "reward_judge_quality_std": 0.22793717682361603, "reward_total_composite_mean": 0.658321738243103, "reward_total_composite_std": 0.14894212782382965} {"timestamp_utc": "2026-04-13T03:00:14Z", "mode": "train", "global_step": 1856, "epoch": 0.18643897538925164, "loss": 0.0073, "grad_norm": 8.208630561828613, "learning_rate": 4.378787878787879e-06, "num_tokens": 3380703.0, "completions/mean_length": 66.5, "completions/min_length": 61.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.8377540111541748, "rewards/meter/std": 0.3260452151298523, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9069114923477173, "rewards/repeat_soft/std": 0.09505674988031387, "rewards/judge_quality/mean": 0.3412500023841858, "rewards/judge_quality/std": 0.12040852755308151, "rewards/total_composite/mean": 0.7200554609298706, "rewards/total_composite/std": 0.17396895587444305, "reward": 0.7200554609298706, "reward_std": 0.17396895587444305, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11912531405687332, "sampling/sampling_logp_difference/max": 2.0833358764648438, "sampling/importance_sampling_ratio/min": 0.1245141550898552, "sampling/importance_sampling_ratio/mean": 0.9976533651351929, "sampling/importance_sampling_ratio/max": 1.9636880159378052, "entropy": 0.7219495922327042, "clip_ratio/low_mean": 0.02766798436641693, "clip_ratio/low_min": 0.02766798436641693, "clip_ratio/high_mean": 0.09228116180747747, "clip_ratio/high_max": 0.09228116180747747, "clip_ratio/region_mean": 0.1199491461738944, "reward_total_mean": 0.7200554609298706, "reward_meter_mean": 0.8377540111541748, "reward_meter_std": 0.3260452151298523, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9069114923477173, "reward_repeat_soft_std": 0.09505674988031387, "reward_judge_quality_mean": 0.3412500023841858, "reward_judge_quality_std": 0.12040852755308151, "reward_total_composite_mean": 0.7200554609298706, "reward_total_composite_std": 0.17396895587444305} {"timestamp_utc": "2026-04-13T03:00:21Z", "mode": "train", "global_step": 1857, "epoch": 0.18653942742340532, "loss": -0.031, "grad_norm": 10.938190460205078, "learning_rate": 4.375757575757576e-06, "num_tokens": 3382332.0, "completions/mean_length": 43.625, "completions/min_length": 35.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.625, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.7903545498847961, "rewards/meter/std": 0.3081667423248291, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9449750185012817, "rewards/repeat_soft/std": 0.046386487782001495, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.7032819986343384, "rewards/total_composite/std": 0.12980471551418304, "reward": 0.7032819986343384, "reward_std": 0.12980471551418304, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10320328176021576, "sampling/sampling_logp_difference/max": 2.5755064487457275, "sampling/importance_sampling_ratio/min": 0.07611526548862457, "sampling/importance_sampling_ratio/mean": 1.0040533542633057, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4224913828074932, "clip_ratio/low_mean": 0.02269736910238862, "clip_ratio/low_min": 0.02269736910238862, "clip_ratio/high_mean": 0.05953921587206423, "clip_ratio/high_max": 0.05953921587206423, "clip_ratio/region_mean": 0.08223658497445285, "reward_total_mean": 0.7032819986343384, "reward_meter_mean": 0.7903545498847961, "reward_meter_std": 0.3081667423248291, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9449750185012817, "reward_repeat_soft_std": 0.046386487782001495, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.7032819986343384, "reward_total_composite_std": 0.12980471551418304} {"timestamp_utc": "2026-04-13T03:00:28Z", "mode": "train", "global_step": 1858, "epoch": 0.18663987945755903, "loss": -0.0491, "grad_norm": 6.223133563995361, "learning_rate": 4.372727272727273e-06, "num_tokens": 3384663.0, "completions/mean_length": 110.375, "completions/min_length": 95.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.375, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.9872196316719055, "rewards/meter/std": 0.008821898140013218, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.766819953918457, "rewards/repeat_soft/std": 0.12884068489074707, "rewards/judge_quality/mean": 0.24250000715255737, "rewards/judge_quality/std": 0.11792854964733124, "rewards/total_composite/mean": 0.6554629802703857, "rewards/total_composite/std": 0.2671525478363037, "reward": 0.6554629802703857, "reward_std": 0.2671525180339813, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06721474975347519, "sampling/sampling_logp_difference/max": 1.9169485569000244, "sampling/importance_sampling_ratio/min": 0.1470550149679184, "sampling/importance_sampling_ratio/mean": 1.0121732950210571, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.32497851736843586, "clip_ratio/low_mean": 0.005263158120214939, "clip_ratio/low_min": 0.005263158120214939, "clip_ratio/high_mean": 0.051595655269920826, "clip_ratio/high_max": 0.051595655269920826, "clip_ratio/region_mean": 0.056858813390135765, "reward_total_mean": 0.6554629802703857, "reward_meter_mean": 0.9872196316719055, "reward_meter_std": 0.008821898140013218, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.766819953918457, "reward_repeat_soft_std": 0.12884068489074707, "reward_judge_quality_mean": 0.24250000715255737, "reward_judge_quality_std": 0.11792854964733124, "reward_total_composite_mean": 0.6554629802703857, "reward_total_composite_std": 0.2671525478363037} {"timestamp_utc": "2026-04-13T03:00:35Z", "mode": "train", "global_step": 1859, "epoch": 0.1867403314917127, "loss": 0.0311, "grad_norm": 11.360483169555664, "learning_rate": 4.36969696969697e-06, "num_tokens": 3386691.0, "completions/mean_length": 64.5, "completions/min_length": 59.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.5, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.8391431570053101, "rewards/meter/std": 0.1468450427055359, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8861798048019409, "rewards/repeat_soft/std": 0.09407883137464523, "rewards/judge_quality/mean": 0.24375000596046448, "rewards/judge_quality/std": 0.0176776684820652, "rewards/total_composite/mean": 0.6893573999404907, "rewards/total_composite/std": 0.06024165824055672, "reward": 0.6893573999404907, "reward_std": 0.06024165451526642, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08219512552022934, "sampling/sampling_logp_difference/max": 2.213350534439087, "sampling/importance_sampling_ratio/min": 0.10933370888233185, "sampling/importance_sampling_ratio/mean": 1.0095386505126953, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.33959166519343853, "clip_ratio/low_mean": 0.022874751593917608, "clip_ratio/low_min": 0.022874751593917608, "clip_ratio/high_mean": 0.037690824596211314, "clip_ratio/high_max": 0.037690824596211314, "clip_ratio/region_mean": 0.06056557619012892, "reward_total_mean": 0.6893573999404907, "reward_meter_mean": 0.8391431570053101, "reward_meter_std": 0.1468450427055359, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8861798048019409, "reward_repeat_soft_std": 0.09407883137464523, "reward_judge_quality_mean": 0.24375000596046448, "reward_judge_quality_std": 0.0176776684820652, "reward_total_composite_mean": 0.6893573999404907, "reward_total_composite_std": 0.06024165824055672} {"timestamp_utc": "2026-04-13T03:00:43Z", "mode": "train", "global_step": 1860, "epoch": 0.1868407835258664, "loss": 0.0134, "grad_norm": 5.259166717529297, "learning_rate": 4.366666666666667e-06, "num_tokens": 3389136.0, "completions/mean_length": 121.625, "completions/min_length": 116.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 121.625, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.8726290464401245, "rewards/meter/std": 0.16406334936618805, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8558422923088074, "rewards/repeat_soft/std": 0.05644090101122856, "rewards/judge_quality/mean": 0.17750000953674316, "rewards/judge_quality/std": 0.09953463077545166, "rewards/total_composite/mean": 0.6815173029899597, "rewards/total_composite/std": 0.08774706721305847, "reward": 0.6815173029899597, "reward_std": 0.08774705976247787, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07753569632768631, "sampling/sampling_logp_difference/max": 1.7647714614868164, "sampling/importance_sampling_ratio/min": 0.1712259203195572, "sampling/importance_sampling_ratio/mean": 1.008474588394165, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4905957654118538, "clip_ratio/low_mean": 0.026390582788735628, "clip_ratio/low_min": 0.026390582788735628, "clip_ratio/high_mean": 0.04865827178582549, "clip_ratio/high_max": 0.04865827178582549, "clip_ratio/region_mean": 0.07504885457456112, "reward_total_mean": 0.6815173029899597, "reward_meter_mean": 0.8726290464401245, "reward_meter_std": 0.16406334936618805, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8558422923088074, "reward_repeat_soft_std": 0.05644090101122856, "reward_judge_quality_mean": 0.17750000953674316, "reward_judge_quality_std": 0.09953463077545166, "reward_total_composite_mean": 0.6815173029899597, "reward_total_composite_std": 0.08774706721305847} {"timestamp_utc": "2026-04-13T03:00:50Z", "mode": "train", "global_step": 1861, "epoch": 0.1869412355600201, "loss": 0.0675, "grad_norm": 10.435943603515625, "learning_rate": 4.363636363636364e-06, "num_tokens": 3391463.0, "completions/mean_length": 112.875, "completions/min_length": 100.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.875, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.9707645177841187, "rewards/meter/std": 0.03471256420016289, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7289034128189087, "rewards/repeat_soft/std": 0.10461926460266113, "rewards/judge_quality/mean": 0.23000000417232513, "rewards/judge_quality/std": 0.12224097549915314, "rewards/total_composite/mean": 0.6404198408126831, "rewards/total_composite/std": 0.2634037137031555, "reward": 0.6404198408126831, "reward_std": 0.2634037137031555, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07933969795703888, "sampling/sampling_logp_difference/max": 2.4591875076293945, "sampling/importance_sampling_ratio/min": 0.08550439774990082, "sampling/importance_sampling_ratio/mean": 1.0004726648330688, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3293581176549196, "clip_ratio/low_mean": 0.009999999776482582, "clip_ratio/low_min": 0.009999999776482582, "clip_ratio/high_mean": 0.07188134640455246, "clip_ratio/high_max": 0.07188134640455246, "clip_ratio/region_mean": 0.08188134618103504, "reward_total_mean": 0.6404198408126831, "reward_meter_mean": 0.9707645177841187, "reward_meter_std": 0.03471256420016289, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7289034128189087, "reward_repeat_soft_std": 0.10461926460266113, "reward_judge_quality_mean": 0.23000000417232513, "reward_judge_quality_std": 0.12224097549915314, "reward_total_composite_mean": 0.6404198408126831, "reward_total_composite_std": 0.2634037137031555} {"timestamp_utc": "2026-04-13T03:00:57Z", "mode": "train", "global_step": 1862, "epoch": 0.18704168759417378, "loss": 0.0308, "grad_norm": 8.713722229003906, "learning_rate": 4.36060606060606e-06, "num_tokens": 3393435.0, "completions/mean_length": 66.5, "completions/min_length": 61.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9864568114280701, "rewards/meter/std": 0.0192946195602417, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624818563461304, "rewards/repeat_soft/std": 0.020212242379784584, "rewards/judge_quality/mean": 0.2887499928474426, "rewards/judge_quality/std": 0.14865466952323914, "rewards/total_composite/mean": 0.7767787575721741, "rewards/total_composite/std": 0.04105447605252266, "reward": 0.7767787575721741, "reward_std": 0.04105447605252266, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11583328992128372, "sampling/sampling_logp_difference/max": 1.3772714138031006, "sampling/importance_sampling_ratio/min": 0.25226595997810364, "sampling/importance_sampling_ratio/mean": 1.0098716020584106, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6705252379179001, "clip_ratio/low_mean": 0.04130844725295901, "clip_ratio/low_min": 0.04130844725295901, "clip_ratio/high_mean": 0.05417110305279493, "clip_ratio/high_max": 0.05417110305279493, "clip_ratio/region_mean": 0.09547955030575395, "reward_total_mean": 0.7767787575721741, "reward_meter_mean": 0.9864568114280701, "reward_meter_std": 0.0192946195602417, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624818563461304, "reward_repeat_soft_std": 0.020212242379784584, "reward_judge_quality_mean": 0.2887499928474426, "reward_judge_quality_std": 0.14865466952323914, "reward_total_composite_mean": 0.7767787575721741, "reward_total_composite_std": 0.04105447605252266} {"timestamp_utc": "2026-04-13T03:01:06Z", "mode": "train", "global_step": 1863, "epoch": 0.1871421396283275, "loss": -0.0133, "grad_norm": 4.687717437744141, "learning_rate": 4.3575757575757576e-06, "num_tokens": 3396840.0, "completions/mean_length": 216.625, "completions/min_length": 190.0, "completions/max_length": 237.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 216.625, "completions/min_terminated_length": 190.0, "completions/max_terminated_length": 237.0, "rewards/meter/mean": 0.8979668617248535, "rewards/meter/std": 0.13374456763267517, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7905303239822388, "rewards/repeat_soft/std": 0.10600098222494125, "rewards/judge_quality/mean": 0.23374998569488525, "rewards/judge_quality/std": 0.09006941318511963, "rewards/total_composite/mean": 0.7001380920410156, "rewards/total_composite/std": 0.06885156780481339, "reward": 0.7001380920410156, "reward_std": 0.06885156035423279, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06414243578910828, "sampling/sampling_logp_difference/max": 1.4058539867401123, "sampling/importance_sampling_ratio/min": 0.24515759944915771, "sampling/importance_sampling_ratio/mean": 1.0070422887802124, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4014236442744732, "clip_ratio/low_mean": 0.03186992951668799, "clip_ratio/low_min": 0.03186992951668799, "clip_ratio/high_mean": 0.04393461253494024, "clip_ratio/high_max": 0.04393461253494024, "clip_ratio/region_mean": 0.07580454205162823, "reward_total_mean": 0.7001380920410156, "reward_meter_mean": 0.8979668617248535, "reward_meter_std": 0.13374456763267517, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7905303239822388, "reward_repeat_soft_std": 0.10600098222494125, "reward_judge_quality_mean": 0.23374998569488525, "reward_judge_quality_std": 0.09006941318511963, "reward_total_composite_mean": 0.7001380920410156, "reward_total_composite_std": 0.06885156780481339} {"timestamp_utc": "2026-04-13T03:01:13Z", "mode": "train", "global_step": 1864, "epoch": 0.18724259166248117, "loss": 0.0468, "grad_norm": 7.2530903816223145, "learning_rate": 4.354545454545455e-06, "num_tokens": 3399031.0, "completions/mean_length": 100.875, "completions/min_length": 93.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.875, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.8159576654434204, "rewards/meter/std": 0.272695928812027, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9281368851661682, "rewards/repeat_soft/std": 0.04197493940591812, "rewards/judge_quality/mean": 0.33500000834465027, "rewards/judge_quality/std": 0.12972497940063477, "rewards/total_composite/mean": 0.7104946374893188, "rewards/total_composite/std": 0.13565270602703094, "reward": 0.7104946374893188, "reward_std": 0.13565269112586975, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10837927460670471, "sampling/sampling_logp_difference/max": 3.2068095207214355, "sampling/importance_sampling_ratio/min": 0.040485575795173645, "sampling/importance_sampling_ratio/mean": 1.008068323135376, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5941840521991253, "clip_ratio/low_mean": 0.04705297760665417, "clip_ratio/low_min": 0.04705297760665417, "clip_ratio/high_mean": 0.06486703176051378, "clip_ratio/high_max": 0.06486703176051378, "clip_ratio/region_mean": 0.11192000936716795, "reward_total_mean": 0.7104946374893188, "reward_meter_mean": 0.8159576654434204, "reward_meter_std": 0.272695928812027, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9281368851661682, "reward_repeat_soft_std": 0.04197493940591812, "reward_judge_quality_mean": 0.33500000834465027, "reward_judge_quality_std": 0.12972497940063477, "reward_total_composite_mean": 0.7104946374893188, "reward_total_composite_std": 0.13565270602703094} {"timestamp_utc": "2026-04-13T03:01:19Z", "mode": "train", "global_step": 1865, "epoch": 0.18734304369663485, "loss": -0.0072, "grad_norm": 15.293724060058594, "learning_rate": 4.351515151515152e-06, "num_tokens": 3400753.0, "completions/mean_length": 42.25, "completions/min_length": 32.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.25, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.7734663486480713, "rewards/meter/std": 0.3103182911872864, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9384551644325256, "rewards/repeat_soft/std": 0.0693584755063057, "rewards/judge_quality/mean": 0.48125001788139343, "rewards/judge_quality/std": 0.2968134582042694, "rewards/total_composite/mean": 0.7362803816795349, "rewards/total_composite/std": 0.15644584596157074, "reward": 0.7362803816795349, "reward_std": 0.15644586086273193, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1018277034163475, "sampling/sampling_logp_difference/max": 1.2440500259399414, "sampling/importance_sampling_ratio/min": 0.2882145643234253, "sampling/importance_sampling_ratio/mean": 1.0097553730010986, "sampling/importance_sampling_ratio/max": 1.7275468111038208, "entropy": 0.5242415182292461, "clip_ratio/low_mean": 0.050662403693422675, "clip_ratio/low_min": 0.050662403693422675, "clip_ratio/high_mean": 0.030296728247776628, "clip_ratio/high_max": 0.030296728247776628, "clip_ratio/region_mean": 0.0809591319411993, "reward_total_mean": 0.7362803816795349, "reward_meter_mean": 0.7734663486480713, "reward_meter_std": 0.3103182911872864, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9384551644325256, "reward_repeat_soft_std": 0.0693584755063057, "reward_judge_quality_mean": 0.48125001788139343, "reward_judge_quality_std": 0.2968134582042694, "reward_total_composite_mean": 0.7362803816795349, "reward_total_composite_std": 0.15644584596157074} {"timestamp_utc": "2026-04-13T03:01:31Z", "mode": "train", "global_step": 1866, "epoch": 0.18744349573078856, "loss": -0.2262, "grad_norm": 1.6419951915740967, "learning_rate": 4.348484848484849e-06, "num_tokens": 3403197.0, "completions/mean_length": 195.5, "completions/min_length": 145.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 150.2857208251953, "completions/min_terminated_length": 145.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.9067345261573792, "rewards/meter/std": 0.2058652639389038, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8608520030975342, "rewards/repeat_soft/std": 0.11373545229434967, "rewards/judge_quality/mean": 0.23000000417232513, "rewards/judge_quality/std": 0.13341663777828217, "rewards/total_composite/mean": 0.6253387928009033, "rewards/total_composite/std": 0.2635655999183655, "reward": 0.6253387928009033, "reward_std": 0.2635655701160431, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08214672654867172, "sampling/sampling_logp_difference/max": 1.6031110286712646, "sampling/importance_sampling_ratio/min": 0.20126938819885254, "sampling/importance_sampling_ratio/mean": 0.9971529245376587, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36303776875138283, "clip_ratio/low_mean": 0.013513513840734959, "clip_ratio/low_min": 0.013513513840734959, "clip_ratio/high_mean": 0.06621731165796518, "clip_ratio/high_max": 0.06621731165796518, "clip_ratio/region_mean": 0.07973082549870014, "reward_total_mean": 0.6253387928009033, "reward_meter_mean": 0.9067345261573792, "reward_meter_std": 0.2058652639389038, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8608520030975342, "reward_repeat_soft_std": 0.11373545229434967, "reward_judge_quality_mean": 0.23000000417232513, "reward_judge_quality_std": 0.13341663777828217, "reward_total_composite_mean": 0.6253387928009033, "reward_total_composite_std": 0.2635655999183655} {"timestamp_utc": "2026-04-13T03:01:43Z", "mode": "train", "global_step": 1867, "epoch": 0.18754394776494224, "loss": -0.1206, "grad_norm": 2.517075300216675, "learning_rate": 4.345454545454546e-06, "num_tokens": 3404988.0, "completions/mean_length": 107.875, "completions/min_length": 44.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 50.142860412597656, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.6290701627731323, "rewards/meter/std": 0.382223904132843, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9486645460128784, "rewards/repeat_soft/std": 0.0353502593934536, "rewards/judge_quality/mean": 0.39499998092651367, "rewards/judge_quality/std": 0.2566264867782593, "rewards/total_composite/mean": 0.6137908101081848, "rewards/total_composite/std": 0.2793295979499817, "reward": 0.6137908101081848, "reward_std": 0.2793295979499817, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11061335355043411, "sampling/sampling_logp_difference/max": 1.889897108078003, "sampling/importance_sampling_ratio/min": 0.15108735859394073, "sampling/importance_sampling_ratio/mean": 1.0009040832519531, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4922695606946945, "clip_ratio/low_mean": 0.026714514009654522, "clip_ratio/low_min": 0.026714514009654522, "clip_ratio/high_mean": 0.08640970848500729, "clip_ratio/high_max": 0.08640970848500729, "clip_ratio/region_mean": 0.11312422249466181, "reward_total_mean": 0.6137908101081848, "reward_meter_mean": 0.6290701627731323, "reward_meter_std": 0.382223904132843, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9486645460128784, "reward_repeat_soft_std": 0.0353502593934536, "reward_judge_quality_mean": 0.39499998092651367, "reward_judge_quality_std": 0.2566264867782593, "reward_total_composite_mean": 0.6137908101081848, "reward_total_composite_std": 0.2793295979499817} {"timestamp_utc": "2026-04-13T03:01:50Z", "mode": "train", "global_step": 1868, "epoch": 0.18764439979909592, "loss": 0.0324, "grad_norm": 10.431429862976074, "learning_rate": 4.342424242424243e-06, "num_tokens": 3406991.0, "completions/mean_length": 69.375, "completions/min_length": 62.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.375, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9507268071174622, "rewards/meter/std": 0.11430530250072479, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.951970636844635, "rewards/repeat_soft/std": 0.03891231119632721, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.7825241088867188, "rewards/total_composite/std": 0.05497746914625168, "reward": 0.7825241088867188, "reward_std": 0.054977454245090485, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09189075231552124, "sampling/sampling_logp_difference/max": 1.1468889713287354, "sampling/importance_sampling_ratio/min": 0.3176233470439911, "sampling/importance_sampling_ratio/mean": 1.0108860731124878, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48063550144433975, "clip_ratio/low_mean": 0.028469837736338377, "clip_ratio/low_min": 0.028469837736338377, "clip_ratio/high_mean": 0.06759812869131565, "clip_ratio/high_max": 0.06759812869131565, "clip_ratio/region_mean": 0.09606796642765403, "reward_total_mean": 0.7825241088867188, "reward_meter_mean": 0.9507268071174622, "reward_meter_std": 0.11430530250072479, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.951970636844635, "reward_repeat_soft_std": 0.03891231119632721, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.7825241088867188, "reward_total_composite_std": 0.05497746914625168} {"timestamp_utc": "2026-04-13T03:01:56Z", "mode": "train", "global_step": 1869, "epoch": 0.18774485183324963, "loss": 0.0296, "grad_norm": 9.540276527404785, "learning_rate": 4.33939393939394e-06, "num_tokens": 3408442.0, "completions/mean_length": 37.375, "completions/min_length": 30.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.375, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.8881993293762207, "rewards/meter/std": 0.27832698822021484, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9190136194229126, "rewards/repeat_soft/std": 0.03816822171211243, "rewards/judge_quality/mean": 0.3425000011920929, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.7443410754203796, "rewards/total_composite/std": 0.1402447372674942, "reward": 0.7443410754203796, "reward_std": 0.1402447372674942, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09662902355194092, "sampling/sampling_logp_difference/max": 1.236680030822754, "sampling/importance_sampling_ratio/min": 0.29034656286239624, "sampling/importance_sampling_ratio/mean": 1.0282284021377563, "sampling/importance_sampling_ratio/max": 1.9313117265701294, "entropy": 0.6520564816892147, "clip_ratio/low_mean": 0.012917201500386, "clip_ratio/low_min": 0.012917201500386, "clip_ratio/high_mean": 0.04366645868867636, "clip_ratio/high_max": 0.04366645868867636, "clip_ratio/region_mean": 0.05658366018906236, "reward_total_mean": 0.7443410754203796, "reward_meter_mean": 0.8881993293762207, "reward_meter_std": 0.27832698822021484, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9190136194229126, "reward_repeat_soft_std": 0.03816822171211243, "reward_judge_quality_mean": 0.3425000011920929, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.7443410754203796, "reward_total_composite_std": 0.1402447372674942} {"timestamp_utc": "2026-04-13T03:02:07Z", "mode": "train", "global_step": 1870, "epoch": 0.1878453038674033, "loss": -0.0753, "grad_norm": 4.76171875, "learning_rate": 4.336363636363637e-06, "num_tokens": 3409882.0, "completions/mean_length": 102.0, "completions/min_length": 38.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 43.42857360839844, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.5043160915374756, "rewards/meter/std": 0.4051293730735779, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.970820426940918, "rewards/repeat_soft/std": 0.03810800239443779, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.3143445551395416, "rewards/total_composite/mean": 0.5323137640953064, "rewards/total_composite/std": 0.2981618046760559, "reward": 0.5323137640953064, "reward_std": 0.2981617748737335, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11887308210134506, "sampling/sampling_logp_difference/max": 1.590291976928711, "sampling/importance_sampling_ratio/min": 0.20386607944965363, "sampling/importance_sampling_ratio/mean": 0.9933451414108276, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5012618452310562, "clip_ratio/low_mean": 0.013044990599155426, "clip_ratio/low_min": 0.013044990599155426, "clip_ratio/high_mean": 0.09382208902388811, "clip_ratio/high_max": 0.09382208902388811, "clip_ratio/region_mean": 0.10686707962304354, "reward_total_mean": 0.5323137640953064, "reward_meter_mean": 0.5043160915374756, "reward_meter_std": 0.4051293730735779, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.970820426940918, "reward_repeat_soft_std": 0.03810800239443779, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.3143445551395416, "reward_total_composite_mean": 0.5323137640953064, "reward_total_composite_std": 0.2981618046760559} {"timestamp_utc": "2026-04-13T03:02:19Z", "mode": "train", "global_step": 1871, "epoch": 0.18794575590155702, "loss": -0.1495, "grad_norm": 1.9688800573349, "learning_rate": 4.333333333333334e-06, "num_tokens": 3411774.0, "completions/mean_length": 127.5, "completions/min_length": 57.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 72.5714340209961, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.8058284521102905, "rewards/meter/std": 0.3437091112136841, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8664072751998901, "rewards/repeat_soft/std": 0.12952858209609985, "rewards/judge_quality/mean": 0.22499999403953552, "rewards/judge_quality/std": 0.14880476891994476, "rewards/total_composite/mean": 0.6338726282119751, "rewards/total_composite/std": 0.26963335275650024, "reward": 0.6338726282119751, "reward_std": 0.26963332295417786, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10861985385417938, "sampling/sampling_logp_difference/max": 1.6389129161834717, "sampling/importance_sampling_ratio/min": 0.19419102370738983, "sampling/importance_sampling_ratio/mean": 0.9966150522232056, "sampling/importance_sampling_ratio/max": 1.8046317100524902, "entropy": 0.5525089278817177, "clip_ratio/low_mean": 0.004999999888241291, "clip_ratio/low_min": 0.004999999888241291, "clip_ratio/high_mean": 0.09053600952029228, "clip_ratio/high_max": 0.09053600952029228, "clip_ratio/region_mean": 0.09553600940853357, "reward_total_mean": 0.6338726282119751, "reward_meter_mean": 0.8058284521102905, "reward_meter_std": 0.3437091112136841, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8664072751998901, "reward_repeat_soft_std": 0.12952858209609985, "reward_judge_quality_mean": 0.22499999403953552, "reward_judge_quality_std": 0.14880476891994476, "reward_total_composite_mean": 0.6338726282119751, "reward_total_composite_std": 0.26963335275650024} {"timestamp_utc": "2026-04-13T03:02:25Z", "mode": "train", "global_step": 1872, "epoch": 0.1880462079357107, "loss": 0.0338, "grad_norm": 10.957444190979004, "learning_rate": 4.330303030303031e-06, "num_tokens": 3413375.0, "completions/mean_length": 30.125, "completions/min_length": 28.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.125, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9915555715560913, "rewards/meter/std": 0.00834928173571825, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9209094047546387, "rewards/repeat_soft/std": 0.060174860060214996, "rewards/judge_quality/mean": 0.32625001668930054, "rewards/judge_quality/std": 0.11350739002227783, "rewards/total_composite/mean": 0.7861659526824951, "rewards/total_composite/std": 0.032520972192287445, "reward": 0.7861659526824951, "reward_std": 0.032520975917577744, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09626898169517517, "sampling/sampling_logp_difference/max": 1.437173843383789, "sampling/importance_sampling_ratio/min": 0.23759829998016357, "sampling/importance_sampling_ratio/mean": 0.999019205570221, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49809911102056503, "clip_ratio/low_mean": 0.05314114689826965, "clip_ratio/low_min": 0.05314114689826965, "clip_ratio/high_mean": 0.0544014279730618, "clip_ratio/high_max": 0.0544014279730618, "clip_ratio/region_mean": 0.10754257487133145, "reward_total_mean": 0.7861659526824951, "reward_meter_mean": 0.9915555715560913, "reward_meter_std": 0.00834928173571825, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9209094047546387, "reward_repeat_soft_std": 0.060174860060214996, "reward_judge_quality_mean": 0.32625001668930054, "reward_judge_quality_std": 0.11350739002227783, "reward_total_composite_mean": 0.7861659526824951, "reward_total_composite_std": 0.032520972192287445} {"timestamp_utc": "2026-04-13T03:02:31Z", "mode": "train", "global_step": 1873, "epoch": 0.18814665996986438, "loss": 0.0305, "grad_norm": 11.800539016723633, "learning_rate": 4.327272727272728e-06, "num_tokens": 3415159.0, "completions/mean_length": 56.0, "completions/min_length": 52.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.0, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9900989532470703, "rewards/meter/std": 0.00399128720164299, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7677894234657288, "rewards/repeat_soft/std": 0.13865461945533752, "rewards/judge_quality/mean": 0.3137499690055847, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7664484977722168, "rewards/total_composite/std": 0.029114391654729843, "reward": 0.7664484977722168, "reward_std": 0.029114389792084694, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07731817662715912, "sampling/sampling_logp_difference/max": 1.682450771331787, "sampling/importance_sampling_ratio/min": 0.18591776490211487, "sampling/importance_sampling_ratio/mean": 1.0179413557052612, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.27069478295743465, "clip_ratio/low_mean": 0.04265870922245085, "clip_ratio/low_min": 0.04265870922245085, "clip_ratio/high_mean": 0.023204378318041563, "clip_ratio/high_max": 0.023204378318041563, "clip_ratio/region_mean": 0.06586308754049242, "reward_total_mean": 0.7664484977722168, "reward_meter_mean": 0.9900989532470703, "reward_meter_std": 0.00399128720164299, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7677894234657288, "reward_repeat_soft_std": 0.13865461945533752, "reward_judge_quality_mean": 0.3137499690055847, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7664484977722168, "reward_total_composite_std": 0.029114391654729843} {"timestamp_utc": "2026-04-13T03:02:39Z", "mode": "train", "global_step": 1874, "epoch": 0.18824711200401809, "loss": 0.0761, "grad_norm": 8.555224418640137, "learning_rate": 4.324242424242425e-06, "num_tokens": 3417561.0, "completions/mean_length": 125.25, "completions/min_length": 109.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.25, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.901947557926178, "rewards/meter/std": 0.14672286808490753, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8147162199020386, "rewards/repeat_soft/std": 0.0945051982998848, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7505980134010315, "rewards/total_composite/std": 0.08197882026433945, "reward": 0.7505980134010315, "reward_std": 0.08197880536317825, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09375610947608948, "sampling/sampling_logp_difference/max": 1.9125280380249023, "sampling/importance_sampling_ratio/min": 0.1477065086364746, "sampling/importance_sampling_ratio/mean": 0.9938874840736389, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43570905178785324, "clip_ratio/low_mean": 0.025502019561827183, "clip_ratio/low_min": 0.025502019561827183, "clip_ratio/high_mean": 0.0629658903926611, "clip_ratio/high_max": 0.0629658903926611, "clip_ratio/region_mean": 0.08846790995448828, "reward_total_mean": 0.7505980134010315, "reward_meter_mean": 0.901947557926178, "reward_meter_std": 0.14672286808490753, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8147162199020386, "reward_repeat_soft_std": 0.0945051982998848, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7505980134010315, "reward_total_composite_std": 0.08197882026433945} {"timestamp_utc": "2026-04-13T03:02:45Z", "mode": "train", "global_step": 1875, "epoch": 0.18834756403817177, "loss": -0.0069, "grad_norm": 13.992069244384766, "learning_rate": 4.321212121212121e-06, "num_tokens": 3419005.0, "completions/mean_length": 30.5, "completions/min_length": 27.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.5, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.8697448372840881, "rewards/meter/std": 0.2174164056777954, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8946386575698853, "rewards/repeat_soft/std": 0.11897114664316177, "rewards/judge_quality/mean": 0.45375001430511475, "rewards/judge_quality/std": 0.3130466639995575, "rewards/total_composite/mean": 0.7669740319252014, "rewards/total_composite/std": 0.1406271904706955, "reward": 0.7669740319252014, "reward_std": 0.1406271904706955, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0841173529624939, "sampling/sampling_logp_difference/max": 2.7411446571350098, "sampling/importance_sampling_ratio/min": 0.0644964799284935, "sampling/importance_sampling_ratio/mean": 1.0062896013259888, "sampling/importance_sampling_ratio/max": 1.6246222257614136, "entropy": 0.5068892948329449, "clip_ratio/low_mean": 0.030126534402370453, "clip_ratio/low_min": 0.030126534402370453, "clip_ratio/high_mean": 0.04734848625957966, "clip_ratio/high_max": 0.04734848625957966, "clip_ratio/region_mean": 0.07747502066195011, "reward_total_mean": 0.7669740319252014, "reward_meter_mean": 0.8697448372840881, "reward_meter_std": 0.2174164056777954, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8946386575698853, "reward_repeat_soft_std": 0.11897114664316177, "reward_judge_quality_mean": 0.45375001430511475, "reward_judge_quality_std": 0.3130466639995575, "reward_total_composite_mean": 0.7669740319252014, "reward_total_composite_std": 0.1406271904706955} {"timestamp_utc": "2026-04-13T03:02:52Z", "mode": "train", "global_step": 1876, "epoch": 0.18844801607232547, "loss": 0.0325, "grad_norm": 5.960237503051758, "learning_rate": 4.3181818181818185e-06, "num_tokens": 3421591.0, "completions/mean_length": 141.25, "completions/min_length": 112.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 141.25, "completions/min_terminated_length": 112.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.8891412019729614, "rewards/meter/std": 0.2949821352958679, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.767662763595581, "rewards/repeat_soft/std": 0.0636095330119133, "rewards/judge_quality/mean": 0.1875, "rewards/judge_quality/std": 0.0517549142241478, "rewards/total_composite/mean": 0.6643798351287842, "rewards/total_composite/std": 0.1353110522031784, "reward": 0.6643798351287842, "reward_std": 0.1353110522031784, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07533884048461914, "sampling/sampling_logp_difference/max": 2.0981407165527344, "sampling/importance_sampling_ratio/min": 0.12268431484699249, "sampling/importance_sampling_ratio/mean": 1.0075942277908325, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.35767483338713646, "clip_ratio/low_mean": 0.005718954373151064, "clip_ratio/low_min": 0.005718954373151064, "clip_ratio/high_mean": 0.058044999837875366, "clip_ratio/high_max": 0.058044999837875366, "clip_ratio/region_mean": 0.06376395421102643, "reward_total_mean": 0.6643798351287842, "reward_meter_mean": 0.8891412019729614, "reward_meter_std": 0.2949821352958679, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.767662763595581, "reward_repeat_soft_std": 0.0636095330119133, "reward_judge_quality_mean": 0.1875, "reward_judge_quality_std": 0.0517549142241478, "reward_total_composite_mean": 0.6643798351287842, "reward_total_composite_std": 0.1353110522031784} {"timestamp_utc": "2026-04-13T03:02:59Z", "mode": "train", "global_step": 1877, "epoch": 0.18854846810647916, "loss": -0.0062, "grad_norm": 5.784326553344727, "learning_rate": 4.315151515151516e-06, "num_tokens": 3423599.0, "completions/mean_length": 82.0, "completions/min_length": 78.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.0, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9869763851165771, "rewards/meter/std": 0.0058296360075473785, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6384367942810059, "rewards/repeat_soft/std": 0.12758374214172363, "rewards/judge_quality/mean": 0.16250000894069672, "rewards/judge_quality/std": 0.0353553369641304, "rewards/total_composite/mean": 0.7067330479621887, "rewards/total_composite/std": 0.0179522093385458, "reward": 0.7067330479621887, "reward_std": 0.01795220375061035, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0687638446688652, "sampling/sampling_logp_difference/max": 1.3721036911010742, "sampling/importance_sampling_ratio/min": 0.253572940826416, "sampling/importance_sampling_ratio/mean": 0.9940357208251953, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2746917996555567, "clip_ratio/low_mean": 0.025077007245272398, "clip_ratio/low_min": 0.025077007245272398, "clip_ratio/high_mean": 0.02777777798473835, "clip_ratio/high_max": 0.02777777798473835, "clip_ratio/region_mean": 0.05285478523001075, "reward_total_mean": 0.7067330479621887, "reward_meter_mean": 0.9869763851165771, "reward_meter_std": 0.0058296360075473785, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6384367942810059, "reward_repeat_soft_std": 0.12758374214172363, "reward_judge_quality_mean": 0.16250000894069672, "reward_judge_quality_std": 0.0353553369641304, "reward_total_composite_mean": 0.7067330479621887, "reward_total_composite_std": 0.0179522093385458} {"timestamp_utc": "2026-04-13T03:03:05Z", "mode": "train", "global_step": 1878, "epoch": 0.18864892014063284, "loss": -0.0123, "grad_norm": 12.115094184875488, "learning_rate": 4.312121212121212e-06, "num_tokens": 3425440.0, "completions/mean_length": 62.125, "completions/min_length": 58.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.125, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.8843796849250793, "rewards/meter/std": 0.15172183513641357, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9101226329803467, "rewards/repeat_soft/std": 0.09767671674489975, "rewards/judge_quality/mean": 0.42124998569488525, "rewards/judge_quality/std": 0.06998724490404129, "rewards/total_composite/mean": 0.7653580904006958, "rewards/total_composite/std": 0.06599138677120209, "reward": 0.7653580904006958, "reward_std": 0.06599139422178268, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10475588589906693, "sampling/sampling_logp_difference/max": 1.5222513675689697, "sampling/importance_sampling_ratio/min": 0.218220055103302, "sampling/importance_sampling_ratio/mean": 0.9968607425689697, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43531327694654465, "clip_ratio/low_mean": 0.018979421816766262, "clip_ratio/low_min": 0.018979421816766262, "clip_ratio/high_mean": 0.08489550231024623, "clip_ratio/high_max": 0.08489550231024623, "clip_ratio/region_mean": 0.10387492412701249, "reward_total_mean": 0.7653580904006958, "reward_meter_mean": 0.8843796849250793, "reward_meter_std": 0.15172183513641357, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9101226329803467, "reward_repeat_soft_std": 0.09767671674489975, "reward_judge_quality_mean": 0.42124998569488525, "reward_judge_quality_std": 0.06998724490404129, "reward_total_composite_mean": 0.7653580904006958, "reward_total_composite_std": 0.06599138677120209} {"timestamp_utc": "2026-04-13T03:03:13Z", "mode": "train", "global_step": 1879, "epoch": 0.18874937217478654, "loss": 0.0015, "grad_norm": 4.989386558532715, "learning_rate": 4.309090909090909e-06, "num_tokens": 3428104.0, "completions/mean_length": 154.0, "completions/min_length": 126.0, "completions/max_length": 165.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 154.0, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 165.0, "rewards/meter/mean": 0.9276673197746277, "rewards/meter/std": 0.1315077394247055, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7188701629638672, "rewards/repeat_soft/std": 0.1013931855559349, "rewards/judge_quality/mean": 0.2762500047683716, "rewards/judge_quality/std": 0.12603145837783813, "rewards/total_composite/mean": 0.7109622955322266, "rewards/total_composite/std": 0.05961180478334427, "reward": 0.7109622955322266, "reward_std": 0.059611789882183075, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06404755264520645, "sampling/sampling_logp_difference/max": 3.503312826156616, "sampling/importance_sampling_ratio/min": 0.030097510665655136, "sampling/importance_sampling_ratio/mean": 1.0026040077209473, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23693015798926353, "clip_ratio/low_mean": 0.030915739480406046, "clip_ratio/low_min": 0.030915739480406046, "clip_ratio/high_mean": 0.027747636195272207, "clip_ratio/high_max": 0.027747636195272207, "clip_ratio/region_mean": 0.05866337567567825, "reward_total_mean": 0.7109622955322266, "reward_meter_mean": 0.9276673197746277, "reward_meter_std": 0.1315077394247055, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7188701629638672, "reward_repeat_soft_std": 0.1013931855559349, "reward_judge_quality_mean": 0.2762500047683716, "reward_judge_quality_std": 0.12603145837783813, "reward_total_composite_mean": 0.7109622955322266, "reward_total_composite_std": 0.05961180478334427} {"timestamp_utc": "2026-04-13T03:03:19Z", "mode": "train", "global_step": 1880, "epoch": 0.18884982420894023, "loss": -0.0114, "grad_norm": 10.338534355163574, "learning_rate": 4.306060606060607e-06, "num_tokens": 3429812.0, "completions/mean_length": 53.5, "completions/min_length": 45.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.5, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9847441911697388, "rewards/meter/std": 0.012587711215019226, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.933705747127533, "rewards/repeat_soft/std": 0.08519002050161362, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.832380473613739, "rewards/total_composite/std": 0.05416525900363922, "reward": 0.832380473613739, "reward_std": 0.05416524037718773, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11218582838773727, "sampling/sampling_logp_difference/max": 1.7283992767333984, "sampling/importance_sampling_ratio/min": 0.17756842076778412, "sampling/importance_sampling_ratio/mean": 1.0046627521514893, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5783138833940029, "clip_ratio/low_mean": 0.04870323999784887, "clip_ratio/low_min": 0.04870323999784887, "clip_ratio/high_mean": 0.022591857239603996, "clip_ratio/high_max": 0.022591857239603996, "clip_ratio/region_mean": 0.07129509723745286, "reward_total_mean": 0.832380473613739, "reward_meter_mean": 0.9847441911697388, "reward_meter_std": 0.012587711215019226, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.933705747127533, "reward_repeat_soft_std": 0.08519002050161362, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.832380473613739, "reward_total_composite_std": 0.05416525900363922} {"timestamp_utc": "2026-04-13T03:03:26Z", "mode": "train", "global_step": 1881, "epoch": 0.18895027624309393, "loss": 0.0451, "grad_norm": 15.708800315856934, "learning_rate": 4.303030303030303e-06, "num_tokens": 3431628.0, "completions/mean_length": 56.0, "completions/min_length": 49.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.0, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9543802738189697, "rewards/meter/std": 0.05629798024892807, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9759899377822876, "rewards/repeat_soft/std": 0.019439799711108208, "rewards/judge_quality/mean": 0.3687499761581421, "rewards/judge_quality/std": 0.186581090092659, "rewards/total_composite/mean": 0.7876951098442078, "rewards/total_composite/std": 0.06552652269601822, "reward": 0.7876951098442078, "reward_std": 0.06552651524543762, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1167442798614502, "sampling/sampling_logp_difference/max": 3.0160417556762695, "sampling/importance_sampling_ratio/min": 0.0489947684109211, "sampling/importance_sampling_ratio/mean": 1.0026311874389648, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6022238656878471, "clip_ratio/low_mean": 0.058809428475797176, "clip_ratio/low_min": 0.058809428475797176, "clip_ratio/high_mean": 0.06834983453154564, "clip_ratio/high_max": 0.06834983453154564, "clip_ratio/region_mean": 0.12715926300734282, "reward_total_mean": 0.7876951098442078, "reward_meter_mean": 0.9543802738189697, "reward_meter_std": 0.05629798024892807, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9759899377822876, "reward_repeat_soft_std": 0.019439799711108208, "reward_judge_quality_mean": 0.3687499761581421, "reward_judge_quality_std": 0.186581090092659, "reward_total_composite_mean": 0.7876951098442078, "reward_total_composite_std": 0.06552652269601822} {"timestamp_utc": "2026-04-13T03:03:38Z", "mode": "train", "global_step": 1882, "epoch": 0.18905072827724761, "loss": -0.0959, "grad_norm": 1.7413051128387451, "learning_rate": 4.3e-06, "num_tokens": 3433149.0, "completions/mean_length": 91.125, "completions/min_length": 27.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 31.000001907348633, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.7989579439163208, "rewards/meter/std": 0.35899657011032104, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9167718887329102, "rewards/repeat_soft/std": 0.06938906759023666, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.6806554198265076, "rewards/total_composite/std": 0.28524863719940186, "reward": 0.6806554198265076, "reward_std": 0.28524863719940186, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09183402359485626, "sampling/sampling_logp_difference/max": 0.9638216495513916, "sampling/importance_sampling_ratio/min": 0.4024847447872162, "sampling/importance_sampling_ratio/mean": 1.0109361410140991, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39346304908394814, "clip_ratio/low_mean": 0.0071428571827709675, "clip_ratio/low_min": 0.0071428571827709675, "clip_ratio/high_mean": 0.06210792763158679, "clip_ratio/high_max": 0.06210792763158679, "clip_ratio/region_mean": 0.06925078481435776, "reward_total_mean": 0.6806554198265076, "reward_meter_mean": 0.7989579439163208, "reward_meter_std": 0.35899657011032104, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9167718887329102, "reward_repeat_soft_std": 0.06938906759023666, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.6806554198265076, "reward_total_composite_std": 0.28524863719940186} {"timestamp_utc": "2026-04-13T03:03:45Z", "mode": "train", "global_step": 1883, "epoch": 0.1891511803114013, "loss": 0.0366, "grad_norm": 6.166489124298096, "learning_rate": 4.296969696969698e-06, "num_tokens": 3435636.0, "completions/mean_length": 142.875, "completions/min_length": 131.0, "completions/max_length": 155.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.875, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 155.0, "rewards/meter/mean": 0.655170738697052, "rewards/meter/std": 0.4160381853580475, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7226429581642151, "rewards/repeat_soft/std": 0.1981745958328247, "rewards/judge_quality/mean": 0.29750001430511475, "rewards/judge_quality/std": 0.1349867582321167, "rewards/total_composite/mean": 0.6063411235809326, "rewards/total_composite/std": 0.19576074182987213, "reward": 0.6063411235809326, "reward_std": 0.19576077163219452, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07291972637176514, "sampling/sampling_logp_difference/max": 1.9997533559799194, "sampling/importance_sampling_ratio/min": 0.13536867499351501, "sampling/importance_sampling_ratio/mean": 1.0027472972869873, "sampling/importance_sampling_ratio/max": 1.8740507364273071, "entropy": 0.39651133865118027, "clip_ratio/low_mean": 0.028300717938691378, "clip_ratio/low_min": 0.028300717938691378, "clip_ratio/high_mean": 0.0397569197230041, "clip_ratio/high_max": 0.0397569197230041, "clip_ratio/region_mean": 0.06805763766169548, "reward_total_mean": 0.6063411235809326, "reward_meter_mean": 0.655170738697052, "reward_meter_std": 0.4160381853580475, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7226429581642151, "reward_repeat_soft_std": 0.1981745958328247, "reward_judge_quality_mean": 0.29750001430511475, "reward_judge_quality_std": 0.1349867582321167, "reward_total_composite_mean": 0.6063411235809326, "reward_total_composite_std": 0.19576074182987213} {"timestamp_utc": "2026-04-13T03:03:52Z", "mode": "train", "global_step": 1884, "epoch": 0.189251632345555, "loss": 0.0281, "grad_norm": 7.787013530731201, "learning_rate": 4.293939393939394e-06, "num_tokens": 3437456.0, "completions/mean_length": 53.5, "completions/min_length": 51.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.5, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.8670626878738403, "rewards/meter/std": 0.19560113549232483, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9475851058959961, "rewards/repeat_soft/std": 0.03622370585799217, "rewards/judge_quality/mean": 0.5600000023841858, "rewards/judge_quality/std": 0.22258226573467255, "rewards/total_composite/mean": 0.8029367327690125, "rewards/total_composite/std": 0.13068023324012756, "reward": 0.8029367327690125, "reward_std": 0.13068021833896637, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08810829371213913, "sampling/sampling_logp_difference/max": 1.3314399719238281, "sampling/importance_sampling_ratio/min": 0.2640967071056366, "sampling/importance_sampling_ratio/mean": 0.9936099648475647, "sampling/importance_sampling_ratio/max": 1.8202166557312012, "entropy": 0.3529294990003109, "clip_ratio/low_mean": 0.03182167513296008, "clip_ratio/low_min": 0.03182167513296008, "clip_ratio/high_mean": 0.06601802352815866, "clip_ratio/high_max": 0.06601802352815866, "clip_ratio/region_mean": 0.09783969866111875, "reward_total_mean": 0.8029367327690125, "reward_meter_mean": 0.8670626878738403, "reward_meter_std": 0.19560113549232483, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9475851058959961, "reward_repeat_soft_std": 0.03622370585799217, "reward_judge_quality_mean": 0.5600000023841858, "reward_judge_quality_std": 0.22258226573467255, "reward_total_composite_mean": 0.8029367327690125, "reward_total_composite_std": 0.13068023324012756} {"timestamp_utc": "2026-04-13T03:03:59Z", "mode": "train", "global_step": 1885, "epoch": 0.18935208437970869, "loss": 0.0384, "grad_norm": 20.65100860595703, "learning_rate": 4.290909090909091e-06, "num_tokens": 3438939.0, "completions/mean_length": 36.375, "completions/min_length": 29.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.375, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.768007755279541, "rewards/meter/std": 0.3858395516872406, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9092690944671631, "rewards/repeat_soft/std": 0.03748049959540367, "rewards/judge_quality/mean": 0.3799999952316284, "rewards/judge_quality/std": 0.11501552164554596, "rewards/total_composite/mean": 0.7005304098129272, "rewards/total_composite/std": 0.16684973239898682, "reward": 0.7005304098129272, "reward_std": 0.16684973239898682, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12730054557323456, "sampling/sampling_logp_difference/max": 1.5892925262451172, "sampling/importance_sampling_ratio/min": 0.20406992733478546, "sampling/importance_sampling_ratio/mean": 1.0162097215652466, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.677179753780365, "clip_ratio/low_mean": 0.028219396248459816, "clip_ratio/low_min": 0.028219396248459816, "clip_ratio/high_mean": 0.12404408026486635, "clip_ratio/high_max": 0.12404408026486635, "clip_ratio/region_mean": 0.15226347651332617, "reward_total_mean": 0.7005304098129272, "reward_meter_mean": 0.768007755279541, "reward_meter_std": 0.3858395516872406, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9092690944671631, "reward_repeat_soft_std": 0.03748049959540367, "reward_judge_quality_mean": 0.3799999952316284, "reward_judge_quality_std": 0.11501552164554596, "reward_total_composite_mean": 0.7005304098129272, "reward_total_composite_std": 0.16684973239898682} {"timestamp_utc": "2026-04-13T03:04:05Z", "mode": "train", "global_step": 1886, "epoch": 0.1894525364138624, "loss": 0.0136, "grad_norm": 9.707982063293457, "learning_rate": 4.287878787878788e-06, "num_tokens": 3440643.0, "completions/mean_length": 43.0, "completions/min_length": 35.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.820433497428894, "rewards/meter/std": 0.2933847904205322, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9699130058288574, "rewards/repeat_soft/std": 0.027293242514133453, "rewards/judge_quality/mean": 0.41749998927116394, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.7414363622665405, "rewards/total_composite/std": 0.13010592758655548, "reward": 0.7414363622665405, "reward_std": 0.13010594248771667, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10660550743341446, "sampling/sampling_logp_difference/max": 1.6506562232971191, "sampling/importance_sampling_ratio/min": 0.19192393124103546, "sampling/importance_sampling_ratio/mean": 0.999479353427887, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4762999452650547, "clip_ratio/low_mean": 0.02579365111887455, "clip_ratio/low_min": 0.02579365111887455, "clip_ratio/high_mean": 0.09858116135001183, "clip_ratio/high_max": 0.09858116135001183, "clip_ratio/region_mean": 0.12437481246888638, "reward_total_mean": 0.7414363622665405, "reward_meter_mean": 0.820433497428894, "reward_meter_std": 0.2933847904205322, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9699130058288574, "reward_repeat_soft_std": 0.027293242514133453, "reward_judge_quality_mean": 0.41749998927116394, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.7414363622665405, "reward_total_composite_std": 0.13010592758655548} {"timestamp_utc": "2026-04-13T03:04:12Z", "mode": "train", "global_step": 1887, "epoch": 0.18955298844801607, "loss": 0.0523, "grad_norm": 9.433754920959473, "learning_rate": 4.284848484848485e-06, "num_tokens": 3442343.0, "completions/mean_length": 67.5, "completions/min_length": 57.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.5, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.7393182516098022, "rewards/meter/std": 0.4311006963253021, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.877156138420105, "rewards/repeat_soft/std": 0.09002260863780975, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.69865882396698, "rewards/total_composite/std": 0.19477838277816772, "reward": 0.69865882396698, "reward_std": 0.19477838277816772, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08933289349079132, "sampling/sampling_logp_difference/max": 2.591188669204712, "sampling/importance_sampling_ratio/min": 0.29084011912345886, "sampling/importance_sampling_ratio/mean": 1.0226017236709595, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5066510736942291, "clip_ratio/low_mean": 0.023268398828804493, "clip_ratio/low_min": 0.023268398828804493, "clip_ratio/high_mean": 0.05149590002838522, "clip_ratio/high_max": 0.05149590002838522, "clip_ratio/region_mean": 0.07476429885718971, "reward_total_mean": 0.69865882396698, "reward_meter_mean": 0.7393182516098022, "reward_meter_std": 0.4311006963253021, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.877156138420105, "reward_repeat_soft_std": 0.09002260863780975, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.69865882396698, "reward_total_composite_std": 0.19477838277816772} {"timestamp_utc": "2026-04-13T03:04:19Z", "mode": "train", "global_step": 1888, "epoch": 0.18965344048216976, "loss": 0.0458, "grad_norm": 7.149600982666016, "learning_rate": 4.281818181818182e-06, "num_tokens": 3444445.0, "completions/mean_length": 100.75, "completions/min_length": 93.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.75, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.9611231684684753, "rewards/meter/std": 0.035102199763059616, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9105839133262634, "rewards/repeat_soft/std": 0.02494504675269127, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7931888103485107, "rewards/total_composite/std": 0.023272639140486717, "reward": 0.7931888103485107, "reward_std": 0.023272627964615822, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07758986204862595, "sampling/sampling_logp_difference/max": 1.5552265644073486, "sampling/importance_sampling_ratio/min": 0.21114154160022736, "sampling/importance_sampling_ratio/mean": 0.9906851053237915, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3382999263703823, "clip_ratio/low_mean": 0.020934308879077435, "clip_ratio/low_min": 0.020934308879077435, "clip_ratio/high_mean": 0.061227974481880665, "clip_ratio/high_max": 0.061227974481880665, "clip_ratio/region_mean": 0.0821622833609581, "reward_total_mean": 0.7931888103485107, "reward_meter_mean": 0.9611231684684753, "reward_meter_std": 0.035102199763059616, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9105839133262634, "reward_repeat_soft_std": 0.02494504675269127, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7931888103485107, "reward_total_composite_std": 0.023272639140486717} {"timestamp_utc": "2026-04-13T03:04:25Z", "mode": "train", "global_step": 1889, "epoch": 0.18975389251632346, "loss": 0.0268, "grad_norm": 18.767345428466797, "learning_rate": 4.278787878787879e-06, "num_tokens": 3446039.0, "completions/mean_length": 37.25, "completions/min_length": 32.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.25, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9061740636825562, "rewards/meter/std": 0.22301238775253296, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8608391880989075, "rewards/repeat_soft/std": 0.05639398843050003, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.263435423374176, "rewards/total_composite/mean": 0.8459872603416443, "rewards/total_composite/std": 0.14714676141738892, "reward": 0.8459872603416443, "reward_std": 0.14714674651622772, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11742830276489258, "sampling/sampling_logp_difference/max": 2.2892658710479736, "sampling/importance_sampling_ratio/min": 0.10134083777666092, "sampling/importance_sampling_ratio/mean": 0.9790909290313721, "sampling/importance_sampling_ratio/max": 1.8573740720748901, "entropy": 0.3266829289495945, "clip_ratio/low_mean": 0.038017595652490854, "clip_ratio/low_min": 0.038017595652490854, "clip_ratio/high_mean": 0.06482669897377491, "clip_ratio/high_max": 0.06482669897377491, "clip_ratio/region_mean": 0.10284429462626576, "reward_total_mean": 0.8459872603416443, "reward_meter_mean": 0.9061740636825562, "reward_meter_std": 0.22301238775253296, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8608391880989075, "reward_repeat_soft_std": 0.05639398843050003, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.263435423374176, "reward_total_composite_mean": 0.8459872603416443, "reward_total_composite_std": 0.14714676141738892} {"timestamp_utc": "2026-04-13T03:04:32Z", "mode": "train", "global_step": 1890, "epoch": 0.18985434455047714, "loss": 0.0449, "grad_norm": 4.696111679077148, "learning_rate": 4.275757575757576e-06, "num_tokens": 3448438.0, "completions/mean_length": 111.875, "completions/min_length": 97.0, "completions/max_length": 127.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.875, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.9861181378364563, "rewards/meter/std": 0.007683973293751478, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6928765177726746, "rewards/repeat_soft/std": 0.12199194729328156, "rewards/judge_quality/mean": 0.2762500047683716, "rewards/judge_quality/std": 0.12603145837783813, "rewards/total_composite/mean": 0.7459157705307007, "rewards/total_composite/std": 0.04075125977396965, "reward": 0.7459157705307007, "reward_std": 0.04075126349925995, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06644535064697266, "sampling/sampling_logp_difference/max": 1.5535469055175781, "sampling/importance_sampling_ratio/min": 0.2114964723587036, "sampling/importance_sampling_ratio/mean": 0.9922593235969543, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2897315099835396, "clip_ratio/low_mean": 0.01850815722718835, "clip_ratio/low_min": 0.01850815722718835, "clip_ratio/high_mean": 0.04257429437711835, "clip_ratio/high_max": 0.04257429437711835, "clip_ratio/region_mean": 0.0610824516043067, "reward_total_mean": 0.7459157705307007, "reward_meter_mean": 0.9861181378364563, "reward_meter_std": 0.007683973293751478, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6928765177726746, "reward_repeat_soft_std": 0.12199194729328156, "reward_judge_quality_mean": 0.2762500047683716, "reward_judge_quality_std": 0.12603145837783813, "reward_total_composite_mean": 0.7459157705307007, "reward_total_composite_std": 0.04075125977396965} {"timestamp_utc": "2026-04-13T03:04:43Z", "mode": "train", "global_step": 1891, "epoch": 0.18995479658463083, "loss": 0.0685, "grad_norm": 14.447916984558105, "learning_rate": 4.272727272727273e-06, "num_tokens": 3450119.0, "completions/mean_length": 44.125, "completions/min_length": 40.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.125, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.5840741395950317, "rewards/meter/std": 0.34122058749198914, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9642744064331055, "rewards/repeat_soft/std": 0.020245661959052086, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.6397607922554016, "rewards/total_composite/std": 0.15222370624542236, "reward": 0.6397607922554016, "reward_std": 0.15222369134426117, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11971501260995865, "sampling/sampling_logp_difference/max": 2.233919858932495, "sampling/importance_sampling_ratio/min": 0.10710775852203369, "sampling/importance_sampling_ratio/mean": 1.0156439542770386, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5714150965213776, "clip_ratio/low_mean": 0.06056455336511135, "clip_ratio/low_min": 0.06056455336511135, "clip_ratio/high_mean": 0.0387269239872694, "clip_ratio/high_max": 0.0387269239872694, "clip_ratio/region_mean": 0.09929147735238075, "reward_total_mean": 0.6397607922554016, "reward_meter_mean": 0.5840741395950317, "reward_meter_std": 0.34122058749198914, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9642744064331055, "reward_repeat_soft_std": 0.020245661959052086, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.6397607922554016, "reward_total_composite_std": 0.15222370624542236} {"timestamp_utc": "2026-04-13T03:04:50Z", "mode": "train", "global_step": 1892, "epoch": 0.19005524861878453, "loss": 0.0316, "grad_norm": 5.910376071929932, "learning_rate": 4.2696969696969695e-06, "num_tokens": 3452183.0, "completions/mean_length": 88.0, "completions/min_length": 81.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 88.0, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.8887321949005127, "rewards/meter/std": 0.26758521795272827, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9534134864807129, "rewards/repeat_soft/std": 0.02496922016143799, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.12351980805397034, "rewards/total_composite/mean": 0.7382708191871643, "rewards/total_composite/std": 0.14370599389076233, "reward": 0.7382708191871643, "reward_std": 0.14370600879192352, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0783172994852066, "sampling/sampling_logp_difference/max": 1.1788430213928223, "sampling/importance_sampling_ratio/min": 0.30763447284698486, "sampling/importance_sampling_ratio/mean": 1.0178862810134888, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.46275562420487404, "clip_ratio/low_mean": 0.03828237019479275, "clip_ratio/low_min": 0.03828237019479275, "clip_ratio/high_mean": 0.051032816991209984, "clip_ratio/high_max": 0.051032816991209984, "clip_ratio/region_mean": 0.08931518718600273, "reward_total_mean": 0.7382708191871643, "reward_meter_mean": 0.8887321949005127, "reward_meter_std": 0.26758521795272827, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9534134864807129, "reward_repeat_soft_std": 0.02496922016143799, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.12351980805397034, "reward_total_composite_mean": 0.7382708191871643, "reward_total_composite_std": 0.14370599389076233} {"timestamp_utc": "2026-04-13T03:04:56Z", "mode": "train", "global_step": 1893, "epoch": 0.19015570065293821, "loss": 0.0283, "grad_norm": 11.849185943603516, "learning_rate": 4.266666666666668e-06, "num_tokens": 3453830.0, "completions/mean_length": 47.875, "completions/min_length": 43.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.875, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.9735181331634521, "rewards/meter/std": 0.01749442145228386, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9287459850311279, "rewards/repeat_soft/std": 0.04933344945311546, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.1865811049938202, "rewards/total_composite/mean": 0.8403327465057373, "rewards/total_composite/std": 0.053970079869031906, "reward": 0.8403327465057373, "reward_std": 0.053970061242580414, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11341868340969086, "sampling/sampling_logp_difference/max": 2.092376470565796, "sampling/importance_sampling_ratio/min": 0.12339354306459427, "sampling/importance_sampling_ratio/mean": 1.0028760433197021, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6000248938798904, "clip_ratio/low_mean": 0.06488037435337901, "clip_ratio/low_min": 0.06488037435337901, "clip_ratio/high_mean": 0.02678978629410267, "clip_ratio/high_max": 0.02678978629410267, "clip_ratio/region_mean": 0.09167016064748168, "reward_total_mean": 0.8403327465057373, "reward_meter_mean": 0.9735181331634521, "reward_meter_std": 0.01749442145228386, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9287459850311279, "reward_repeat_soft_std": 0.04933344945311546, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.1865811049938202, "reward_total_composite_mean": 0.8403327465057373, "reward_total_composite_std": 0.053970079869031906} {"timestamp_utc": "2026-04-13T03:05:04Z", "mode": "train", "global_step": 1894, "epoch": 0.19025615268709192, "loss": -0.0553, "grad_norm": 5.7205352783203125, "learning_rate": 4.263636363636364e-06, "num_tokens": 3456193.0, "completions/mean_length": 127.375, "completions/min_length": 114.0, "completions/max_length": 153.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.375, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 153.0, "rewards/meter/mean": 0.9946795701980591, "rewards/meter/std": 0.0030092988163232803, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8215070962905884, "rewards/repeat_soft/std": 0.04201636090874672, "rewards/judge_quality/mean": 0.30124998092651367, "rewards/judge_quality/std": 0.10398317128419876, "rewards/total_composite/mean": 0.746694028377533, "rewards/total_composite/std": 0.044930096715688705, "reward": 0.746694028377533, "reward_std": 0.04493008926510811, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07637730240821838, "sampling/sampling_logp_difference/max": 2.6113855838775635, "sampling/importance_sampling_ratio/min": 0.07343272864818573, "sampling/importance_sampling_ratio/mean": 1.0010682344436646, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.34114914759993553, "clip_ratio/low_mean": 0.02563808485865593, "clip_ratio/low_min": 0.02563808485865593, "clip_ratio/high_mean": 0.03971978882327676, "clip_ratio/high_max": 0.03971978882327676, "clip_ratio/region_mean": 0.06535787368193269, "reward_total_mean": 0.746694028377533, "reward_meter_mean": 0.9946795701980591, "reward_meter_std": 0.0030092988163232803, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8215070962905884, "reward_repeat_soft_std": 0.04201636090874672, "reward_judge_quality_mean": 0.30124998092651367, "reward_judge_quality_std": 0.10398317128419876, "reward_total_composite_mean": 0.746694028377533, "reward_total_composite_std": 0.044930096715688705} {"timestamp_utc": "2026-04-13T03:05:10Z", "mode": "train", "global_step": 1895, "epoch": 0.1903566047212456, "loss": 0.0333, "grad_norm": 10.709613800048828, "learning_rate": 4.260606060606061e-06, "num_tokens": 3458212.0, "completions/mean_length": 86.375, "completions/min_length": 74.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.375, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.8109381198883057, "rewards/meter/std": 0.19723083078861237, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8848092555999756, "rewards/repeat_soft/std": 0.05992240458726883, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.700153112411499, "rewards/total_composite/std": 0.0935388132929802, "reward": 0.700153112411499, "reward_std": 0.09353882819414139, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10529804229736328, "sampling/sampling_logp_difference/max": 2.499354124069214, "sampling/importance_sampling_ratio/min": 0.08213803172111511, "sampling/importance_sampling_ratio/mean": 0.9967209696769714, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.38607826083898544, "clip_ratio/low_mean": 0.027309083845466375, "clip_ratio/low_min": 0.027309083845466375, "clip_ratio/high_mean": 0.06424864521250129, "clip_ratio/high_max": 0.06424864521250129, "clip_ratio/region_mean": 0.09155772905796766, "reward_total_mean": 0.700153112411499, "reward_meter_mean": 0.8109381198883057, "reward_meter_std": 0.19723083078861237, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8848092555999756, "reward_repeat_soft_std": 0.05992240458726883, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.700153112411499, "reward_total_composite_std": 0.0935388132929802} {"timestamp_utc": "2026-04-13T03:05:16Z", "mode": "train", "global_step": 1896, "epoch": 0.19045705675539928, "loss": -0.0242, "grad_norm": 14.27367115020752, "learning_rate": 4.2575757575757585e-06, "num_tokens": 3459926.0, "completions/mean_length": 52.25, "completions/min_length": 44.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.25, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.8757286667823792, "rewards/meter/std": 0.2888166010379791, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9248530864715576, "rewards/repeat_soft/std": 0.05603524670004845, "rewards/judge_quality/mean": 0.38874998688697815, "rewards/judge_quality/std": 0.08675704896450043, "rewards/total_composite/mean": 0.7531881928443909, "rewards/total_composite/std": 0.1280779093503952, "reward": 0.7531881928443909, "reward_std": 0.1280779093503952, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12059533596038818, "sampling/sampling_logp_difference/max": 1.565056562423706, "sampling/importance_sampling_ratio/min": 0.2090761810541153, "sampling/importance_sampling_ratio/mean": 1.0026917457580566, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.492951475083828, "clip_ratio/low_mean": 0.011363636702299118, "clip_ratio/low_min": 0.011363636702299118, "clip_ratio/high_mean": 0.09392952313646674, "clip_ratio/high_max": 0.09392952313646674, "clip_ratio/region_mean": 0.10529315983876586, "reward_total_mean": 0.7531881928443909, "reward_meter_mean": 0.8757286667823792, "reward_meter_std": 0.2888166010379791, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9248530864715576, "reward_repeat_soft_std": 0.05603524670004845, "reward_judge_quality_mean": 0.38874998688697815, "reward_judge_quality_std": 0.08675704896450043, "reward_total_composite_mean": 0.7531881928443909, "reward_total_composite_std": 0.1280779093503952} {"timestamp_utc": "2026-04-13T03:05:23Z", "mode": "train", "global_step": 1897, "epoch": 0.190557508789553, "loss": 0.0579, "grad_norm": 14.36962604522705, "learning_rate": 4.254545454545455e-06, "num_tokens": 3461325.0, "completions/mean_length": 34.875, "completions/min_length": 30.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.875, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9874305725097656, "rewards/meter/std": 0.010683352127671242, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9175603985786438, "rewards/repeat_soft/std": 0.05562620982527733, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.8353497982025146, "rewards/total_composite/std": 0.049770623445510864, "reward": 0.8353497982025146, "reward_std": 0.049770623445510864, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.127728670835495, "sampling/sampling_logp_difference/max": 1.9775915145874023, "sampling/importance_sampling_ratio/min": 0.13840216398239136, "sampling/importance_sampling_ratio/mean": 1.0015171766281128, "sampling/importance_sampling_ratio/max": 1.7473762035369873, "entropy": 0.6733145974576473, "clip_ratio/low_mean": 0.06277229357510805, "clip_ratio/low_min": 0.06277229357510805, "clip_ratio/high_mean": 0.01953125, "clip_ratio/high_max": 0.01953125, "clip_ratio/region_mean": 0.08230354357510805, "reward_total_mean": 0.8353497982025146, "reward_meter_mean": 0.9874305725097656, "reward_meter_std": 0.010683352127671242, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9175603985786438, "reward_repeat_soft_std": 0.05562620982527733, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.8353497982025146, "reward_total_composite_std": 0.049770623445510864} {"timestamp_utc": "2026-04-13T03:05:31Z", "mode": "train", "global_step": 1898, "epoch": 0.19065796082370667, "loss": 0.0411, "grad_norm": 3.7999911308288574, "learning_rate": 4.251515151515152e-06, "num_tokens": 3464313.0, "completions/mean_length": 185.5, "completions/min_length": 165.0, "completions/max_length": 208.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 185.5, "completions/min_terminated_length": 165.0, "completions/max_terminated_length": 208.0, "rewards/meter/mean": 0.8174379467964172, "rewards/meter/std": 0.302687406539917, "rewards/count_adherence/mean": 0.8958333134651184, "rewards/count_adherence/std": 0.08625820279121399, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8538858890533447, "rewards/repeat_soft/std": 0.0997610092163086, "rewards/judge_quality/mean": 0.2549999952316284, "rewards/judge_quality/std": 0.1118672639131546, "rewards/total_composite/mean": 0.6641106605529785, "rewards/total_composite/std": 0.14079062640666962, "reward": 0.6641106605529785, "reward_std": 0.14079061150550842, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09258019179105759, "sampling/sampling_logp_difference/max": 3.5503296852111816, "sampling/importance_sampling_ratio/min": 0.02871517278254032, "sampling/importance_sampling_ratio/mean": 1.0060441493988037, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.38366241566836834, "clip_ratio/low_mean": 0.0165064106695354, "clip_ratio/low_min": 0.0165064106695354, "clip_ratio/high_mean": 0.06871848553419113, "clip_ratio/high_max": 0.06871848553419113, "clip_ratio/region_mean": 0.08522489620372653, "reward_total_mean": 0.6641106605529785, "reward_meter_mean": 0.8174379467964172, "reward_meter_std": 0.302687406539917, "reward_count_adherence_mean": 0.8958333134651184, "reward_count_adherence_std": 0.08625820279121399, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8538858890533447, "reward_repeat_soft_std": 0.0997610092163086, "reward_judge_quality_mean": 0.2549999952316284, "reward_judge_quality_std": 0.1118672639131546, "reward_total_composite_mean": 0.6641106605529785, "reward_total_composite_std": 0.14079062640666962} {"timestamp_utc": "2026-04-13T03:05:38Z", "mode": "train", "global_step": 1899, "epoch": 0.19075841285786038, "loss": -0.026, "grad_norm": 8.203167915344238, "learning_rate": 4.248484848484849e-06, "num_tokens": 3465887.0, "completions/mean_length": 52.75, "completions/min_length": 45.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.75, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9342211484909058, "rewards/meter/std": 0.15259352326393127, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8038327693939209, "rewards/repeat_soft/std": 0.08167517930269241, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.779032826423645, "rewards/total_composite/std": 0.06997768580913544, "reward": 0.779032826423645, "reward_std": 0.06997770071029663, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10886785387992859, "sampling/sampling_logp_difference/max": 1.841736078262329, "sampling/importance_sampling_ratio/min": 0.1585419476032257, "sampling/importance_sampling_ratio/mean": 1.0152404308319092, "sampling/importance_sampling_ratio/max": 1.8930182456970215, "entropy": 0.5367934219539165, "clip_ratio/low_mean": 0.011111111380159855, "clip_ratio/low_min": 0.011111111380159855, "clip_ratio/high_mean": 0.08891660626977682, "clip_ratio/high_max": 0.08891660626977682, "clip_ratio/region_mean": 0.10002771764993668, "reward_total_mean": 0.779032826423645, "reward_meter_mean": 0.9342211484909058, "reward_meter_std": 0.15259352326393127, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8038327693939209, "reward_repeat_soft_std": 0.08167517930269241, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.779032826423645, "reward_total_composite_std": 0.06997768580913544} {"timestamp_utc": "2026-04-13T03:05:45Z", "mode": "train", "global_step": 1900, "epoch": 0.19085886489201406, "loss": 0.0165, "grad_norm": 9.09363842010498, "learning_rate": 4.245454545454546e-06, "num_tokens": 3467610.0, "completions/mean_length": 61.375, "completions/min_length": 54.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.375, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9115204811096191, "rewards/meter/std": 0.16636358201503754, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9543931484222412, "rewards/repeat_soft/std": 0.03908344358205795, "rewards/judge_quality/mean": 0.6025000214576721, "rewards/judge_quality/std": 0.35443115234375, "rewards/total_composite/mean": 0.8363735675811768, "rewards/total_composite/std": 0.11118005961179733, "reward": 0.8363735675811768, "reward_std": 0.11118004471063614, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1221221387386322, "sampling/sampling_logp_difference/max": 1.387908935546875, "sampling/importance_sampling_ratio/min": 0.24959668517112732, "sampling/importance_sampling_ratio/mean": 1.0196623802185059, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7893078103661537, "clip_ratio/low_mean": 0.09805373009294271, "clip_ratio/low_min": 0.09805373009294271, "clip_ratio/high_mean": 0.06117510423064232, "clip_ratio/high_max": 0.06117510423064232, "clip_ratio/region_mean": 0.15922883432358503, "reward_total_mean": 0.8363735675811768, "reward_meter_mean": 0.9115204811096191, "reward_meter_std": 0.16636358201503754, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9543931484222412, "reward_repeat_soft_std": 0.03908344358205795, "reward_judge_quality_mean": 0.6025000214576721, "reward_judge_quality_std": 0.35443115234375, "reward_total_composite_mean": 0.8363735675811768, "reward_total_composite_std": 0.11118005961179733} {"timestamp_utc": "2026-04-13T03:06:35Z", "mode": "eval", "global_step": 1900, "epoch": 0.19085886489201406, "eval_loss": NaN, "eval_runtime": 49.6022, "eval_samples_per_second": 1.613, "eval_steps_per_second": 0.202, "eval_num_tokens": 3467610.0, "eval_completions/mean_length": 102.8, "eval_completions/min_length": 45.6, "eval_completions/max_length": 198.1, "eval_completions/clipped_ratio": 0.0125, "eval_completions/mean_terminated_length": 97.83750076293946, "eval_completions/min_terminated_length": 45.6, "eval_completions/max_terminated_length": 164.1, "eval_rewards/meter/mean": 0.8621738493442536, "eval_rewards/meter/std": 0.2287090527359396, "eval_rewards/count_adherence/mean": 0.956249988079071, "eval_rewards/count_adherence/std": 0.09214595891535282, "eval_rewards/hard_gate/mean": 0.9875, "eval_rewards/hard_gate/std": 0.03535533845424652, "eval_rewards/repeat_soft/mean": 0.8851100683212281, "eval_rewards/repeat_soft/std": 0.09421012401580811, "eval_rewards/judge_quality/mean": 0.333624991774559, "eval_rewards/judge_quality/std": 0.14672766625881195, "eval_rewards/total_composite/mean": 0.7185767292976379, "eval_rewards/total_composite/std": 0.12990325465798377, "eval_reward": 0.7185767292976379, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.04353394769132137, "eval_sampling/sampling_logp_difference/max": 0.8400780558586121, "eval_sampling/importance_sampling_ratio/min": 0.44211202263832095, "eval_sampling/importance_sampling_ratio/mean": 1.0079915404319764, "eval_sampling/importance_sampling_ratio/max": 1.316960346698761, "eval_entropy": 0.45777017772197726, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7185767292976379, "eval_reward_meter_mean": 0.8621738493442536, "eval_reward_meter_std": 0.2287090527359396, "eval_reward_count_adherence_mean": 0.956249988079071, "eval_reward_count_adherence_std": 0.09214595891535282, "eval_reward_hard_gate_mean": 0.9875, "eval_reward_hard_gate_std": 0.03535533845424652, "eval_reward_repeat_soft_mean": 0.8851100683212281, "eval_reward_repeat_soft_std": 0.09421012401580811, "eval_reward_judge_quality_mean": 0.333624991774559, "eval_reward_judge_quality_std": 0.14672766625881195, "eval_reward_total_composite_mean": 0.7185767292976379, "eval_reward_total_composite_std": 0.12990325465798377} {"timestamp_utc": "2026-04-13T03:06:44Z", "mode": "train", "global_step": 1901, "epoch": 0.19095931692616774, "loss": 0.0468, "grad_norm": 9.93302059173584, "learning_rate": 4.242424242424243e-06, "num_tokens": 3469334.0, "completions/mean_length": 59.5, "completions/min_length": 53.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.5, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.8590254783630371, "rewards/meter/std": 0.3420616388320923, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9638526439666748, "rewards/repeat_soft/std": 0.039156652987003326, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7600716948509216, "rewards/total_composite/std": 0.15350672602653503, "reward": 0.7600716948509216, "reward_std": 0.15350674092769623, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11248565465211868, "sampling/sampling_logp_difference/max": 1.6248624324798584, "sampling/importance_sampling_ratio/min": 0.19693876802921295, "sampling/importance_sampling_ratio/mean": 1.0044224262237549, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6200861595571041, "clip_ratio/low_mean": 0.02500000037252903, "clip_ratio/low_min": 0.02500000037252903, "clip_ratio/high_mean": 0.10020560771226883, "clip_ratio/high_max": 0.10020560771226883, "clip_ratio/region_mean": 0.12520560808479786, "reward_total_mean": 0.7600716948509216, "reward_meter_mean": 0.8590254783630371, "reward_meter_std": 0.3420616388320923, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9638526439666748, "reward_repeat_soft_std": 0.039156652987003326, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7600716948509216, "reward_total_composite_std": 0.15350672602653503} {"timestamp_utc": "2026-04-13T03:06:51Z", "mode": "train", "global_step": 1902, "epoch": 0.19105976896032145, "loss": -0.0298, "grad_norm": 11.406764030456543, "learning_rate": 4.2393939393939395e-06, "num_tokens": 3470816.0, "completions/mean_length": 33.25, "completions/min_length": 31.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.25, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.7692373991012573, "rewards/meter/std": 0.34697139263153076, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.854915201663971, "rewards/repeat_soft/std": 0.06340406835079193, "rewards/judge_quality/mean": 0.6262500286102295, "rewards/judge_quality/std": 0.2432481348514557, "rewards/total_composite/mean": 0.7695233821868896, "rewards/total_composite/std": 0.14915058016777039, "reward": 0.7695233821868896, "reward_std": 0.14915058016777039, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08887137472629547, "sampling/sampling_logp_difference/max": 1.8748531341552734, "sampling/importance_sampling_ratio/min": 0.1533774882555008, "sampling/importance_sampling_ratio/mean": 1.0191115140914917, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5094783082604408, "clip_ratio/low_mean": 0.02734375, "clip_ratio/low_min": 0.02734375, "clip_ratio/high_mean": 0.06666585663333535, "clip_ratio/high_max": 0.06666585663333535, "clip_ratio/region_mean": 0.09400960663333535, "reward_total_mean": 0.7695233821868896, "reward_meter_mean": 0.7692373991012573, "reward_meter_std": 0.34697139263153076, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.854915201663971, "reward_repeat_soft_std": 0.06340406835079193, "reward_judge_quality_mean": 0.6262500286102295, "reward_judge_quality_std": 0.2432481348514557, "reward_total_composite_mean": 0.7695233821868896, "reward_total_composite_std": 0.14915058016777039} {"timestamp_utc": "2026-04-13T03:06:57Z", "mode": "train", "global_step": 1903, "epoch": 0.19116022099447513, "loss": 0.0323, "grad_norm": 12.299165725708008, "learning_rate": 4.236363636363637e-06, "num_tokens": 3472491.0, "completions/mean_length": 36.375, "completions/min_length": 30.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.375, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.8586423397064209, "rewards/meter/std": 0.1782035082578659, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9229821562767029, "rewards/repeat_soft/std": 0.050490137189626694, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7546872496604919, "rewards/total_composite/std": 0.08336538821458817, "reward": 0.7546872496604919, "reward_std": 0.08336539566516876, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09834761172533035, "sampling/sampling_logp_difference/max": 1.1861543655395508, "sampling/importance_sampling_ratio/min": 0.30539342761039734, "sampling/importance_sampling_ratio/mean": 1.0119634866714478, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49001621827483177, "clip_ratio/low_mean": 0.02329302951693535, "clip_ratio/low_min": 0.02329302951693535, "clip_ratio/high_mean": 0.07263224851340055, "clip_ratio/high_max": 0.07263224851340055, "clip_ratio/region_mean": 0.0959252780303359, "reward_total_mean": 0.7546872496604919, "reward_meter_mean": 0.8586423397064209, "reward_meter_std": 0.1782035082578659, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9229821562767029, "reward_repeat_soft_std": 0.050490137189626694, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7546872496604919, "reward_total_composite_std": 0.08336538821458817} {"timestamp_utc": "2026-04-13T03:07:04Z", "mode": "train", "global_step": 1904, "epoch": 0.19126067302862884, "loss": 0.07, "grad_norm": 12.189836502075195, "learning_rate": 4.233333333333334e-06, "num_tokens": 3474394.0, "completions/mean_length": 67.875, "completions/min_length": 59.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.875, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9880143404006958, "rewards/meter/std": 0.006429567001760006, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9693238735198975, "rewards/repeat_soft/std": 0.022651007398962975, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.8134137988090515, "rewards/total_composite/std": 0.021107885986566544, "reward": 0.8134137988090515, "reward_std": 0.02110786736011505, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10442884266376495, "sampling/sampling_logp_difference/max": 1.5848689079284668, "sampling/importance_sampling_ratio/min": 0.20497466623783112, "sampling/importance_sampling_ratio/mean": 1.015525460243225, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5184279754757881, "clip_ratio/low_mean": 0.02371794916689396, "clip_ratio/low_min": 0.02371794916689396, "clip_ratio/high_mean": 0.08270015008747578, "clip_ratio/high_max": 0.08270015008747578, "clip_ratio/region_mean": 0.10641809925436974, "reward_total_mean": 0.8134137988090515, "reward_meter_mean": 0.9880143404006958, "reward_meter_std": 0.006429567001760006, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9693238735198975, "reward_repeat_soft_std": 0.022651007398962975, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.8134137988090515, "reward_total_composite_std": 0.021107885986566544} {"timestamp_utc": "2026-04-13T03:07:10Z", "mode": "train", "global_step": 1905, "epoch": 0.19136112506278252, "loss": 0.0313, "grad_norm": 10.946322441101074, "learning_rate": 4.2303030303030304e-06, "num_tokens": 3476080.0, "completions/mean_length": 56.75, "completions/min_length": 54.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.75, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9691036939620972, "rewards/meter/std": 0.03746628016233444, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9337923526763916, "rewards/repeat_soft/std": 0.055369969457387924, "rewards/judge_quality/mean": 0.35999998450279236, "rewards/judge_quality/std": 0.09165150672197342, "rewards/total_composite/mean": 0.7874758839607239, "rewards/total_composite/std": 0.030495235696434975, "reward": 0.7874758839607239, "reward_std": 0.030495233833789825, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10989135503768921, "sampling/sampling_logp_difference/max": 1.235149621963501, "sampling/importance_sampling_ratio/min": 0.29079124331474304, "sampling/importance_sampling_ratio/mean": 0.9975699186325073, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5214202739298344, "clip_ratio/low_mean": 0.05595500487834215, "clip_ratio/low_min": 0.05595500487834215, "clip_ratio/high_mean": 0.03699190355837345, "clip_ratio/high_max": 0.03699190355837345, "clip_ratio/region_mean": 0.0929469084367156, "reward_total_mean": 0.7874758839607239, "reward_meter_mean": 0.9691036939620972, "reward_meter_std": 0.03746628016233444, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9337923526763916, "reward_repeat_soft_std": 0.055369969457387924, "reward_judge_quality_mean": 0.35999998450279236, "reward_judge_quality_std": 0.09165150672197342, "reward_total_composite_mean": 0.7874758839607239, "reward_total_composite_std": 0.030495235696434975} {"timestamp_utc": "2026-04-13T03:07:16Z", "mode": "train", "global_step": 1906, "epoch": 0.1914615770969362, "loss": 0.0307, "grad_norm": 12.201725959777832, "learning_rate": 4.227272727272728e-06, "num_tokens": 3477731.0, "completions/mean_length": 57.375, "completions/min_length": 52.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.375, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9829587936401367, "rewards/meter/std": 0.008418295532464981, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.887858510017395, "rewards/repeat_soft/std": 0.09624318778514862, "rewards/judge_quality/mean": 0.3837500214576721, "rewards/judge_quality/std": 0.25235676765441895, "rewards/total_composite/mean": 0.7962422966957092, "rewards/total_composite/std": 0.08344615250825882, "reward": 0.7962422966957092, "reward_std": 0.08344614505767822, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11060727387666702, "sampling/sampling_logp_difference/max": 1.2936205863952637, "sampling/importance_sampling_ratio/min": 0.27427592873573303, "sampling/importance_sampling_ratio/mean": 0.9934254288673401, "sampling/importance_sampling_ratio/max": 1.7141470909118652, "entropy": 0.5534116141498089, "clip_ratio/low_mean": 0.046959681902080774, "clip_ratio/low_min": 0.046959681902080774, "clip_ratio/high_mean": 0.05033540027216077, "clip_ratio/high_max": 0.05033540027216077, "clip_ratio/region_mean": 0.09729508217424154, "reward_total_mean": 0.7962422966957092, "reward_meter_mean": 0.9829587936401367, "reward_meter_std": 0.008418295532464981, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.887858510017395, "reward_repeat_soft_std": 0.09624318778514862, "reward_judge_quality_mean": 0.3837500214576721, "reward_judge_quality_std": 0.25235676765441895, "reward_total_composite_mean": 0.7962422966957092, "reward_total_composite_std": 0.08344615250825882} {"timestamp_utc": "2026-04-13T03:07:22Z", "mode": "train", "global_step": 1907, "epoch": 0.1915620291310899, "loss": -0.0443, "grad_norm": 15.470565795898438, "learning_rate": 4.224242424242425e-06, "num_tokens": 3479156.0, "completions/mean_length": 33.125, "completions/min_length": 28.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.125, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.8836895227432251, "rewards/meter/std": 0.3114084005355835, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9608622789382935, "rewards/repeat_soft/std": 0.0046321069821715355, "rewards/judge_quality/mean": 0.38874998688697815, "rewards/judge_quality/std": 0.08675704896450043, "rewards/total_composite/mean": 0.7603715658187866, "rewards/total_composite/std": 0.15861427783966064, "reward": 0.7603715658187866, "reward_std": 0.15861427783966064, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14162388443946838, "sampling/sampling_logp_difference/max": 1.4214484691619873, "sampling/importance_sampling_ratio/min": 0.24136416614055634, "sampling/importance_sampling_ratio/mean": 1.0050300359725952, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8804747015237808, "clip_ratio/low_mean": 0.008928571827709675, "clip_ratio/low_min": 0.008928571827709675, "clip_ratio/high_mean": 0.1270658578723669, "clip_ratio/high_max": 0.1270658578723669, "clip_ratio/region_mean": 0.13599442970007658, "reward_total_mean": 0.7603715658187866, "reward_meter_mean": 0.8836895227432251, "reward_meter_std": 0.3114084005355835, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9608622789382935, "reward_repeat_soft_std": 0.0046321069821715355, "reward_judge_quality_mean": 0.38874998688697815, "reward_judge_quality_std": 0.08675704896450043, "reward_total_composite_mean": 0.7603715658187866, "reward_total_composite_std": 0.15861427783966064} {"timestamp_utc": "2026-04-13T03:07:34Z", "mode": "train", "global_step": 1908, "epoch": 0.1916624811652436, "loss": -0.1595, "grad_norm": 2.8963868618011475, "learning_rate": 4.221212121212121e-06, "num_tokens": 3481213.0, "completions/mean_length": 147.125, "completions/min_length": 83.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 95.00000762939453, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.6883891820907593, "rewards/meter/std": 0.4111936390399933, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9694766998291016, "rewards/repeat_soft/std": 0.022459011524915695, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.635722815990448, "rewards/total_composite/std": 0.29084011912345886, "reward": 0.635722815990448, "reward_std": 0.29084011912345886, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12962913513183594, "sampling/sampling_logp_difference/max": 3.725679874420166, "sampling/importance_sampling_ratio/min": 0.024096712470054626, "sampling/importance_sampling_ratio/mean": 0.9909052848815918, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45931174978613853, "clip_ratio/low_mean": 0.026606781408190727, "clip_ratio/low_min": 0.026606781408190727, "clip_ratio/high_mean": 0.08643228933215141, "clip_ratio/high_max": 0.08643228933215141, "clip_ratio/region_mean": 0.11303907074034214, "reward_total_mean": 0.635722815990448, "reward_meter_mean": 0.6883891820907593, "reward_meter_std": 0.4111936390399933, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9694766998291016, "reward_repeat_soft_std": 0.022459011524915695, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.635722815990448, "reward_total_composite_std": 0.29084011912345886} {"timestamp_utc": "2026-04-13T03:07:46Z", "mode": "train", "global_step": 1909, "epoch": 0.1917629331993973, "loss": -0.1601, "grad_norm": 1.8285412788391113, "learning_rate": 4.218181818181819e-06, "num_tokens": 3483010.0, "completions/mean_length": 125.625, "completions/min_length": 58.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 70.42857360839844, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.8417487144470215, "rewards/meter/std": 0.3469473123550415, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.919312059879303, "rewards/repeat_soft/std": 0.08353695273399353, "rewards/judge_quality/mean": 0.26249998807907104, "rewards/judge_quality/std": 0.1642080545425415, "rewards/total_composite/mean": 0.6668390035629272, "rewards/total_composite/std": 0.2756544351577759, "reward": 0.6668390035629272, "reward_std": 0.2756544351577759, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10843510180711746, "sampling/sampling_logp_difference/max": 1.6956053972244263, "sampling/importance_sampling_ratio/min": 0.18348811566829681, "sampling/importance_sampling_ratio/mean": 1.0025348663330078, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5482369959354401, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10687681357376277, "clip_ratio/high_max": 0.10687681357376277, "clip_ratio/region_mean": 0.10687681357376277, "reward_total_mean": 0.6668390035629272, "reward_meter_mean": 0.8417487144470215, "reward_meter_std": 0.3469473123550415, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.919312059879303, "reward_repeat_soft_std": 0.08353695273399353, "reward_judge_quality_mean": 0.26249998807907104, "reward_judge_quality_std": 0.1642080545425415, "reward_total_composite_mean": 0.6668390035629272, "reward_total_composite_std": 0.2756544351577759} {"timestamp_utc": "2026-04-13T03:07:53Z", "mode": "train", "global_step": 1910, "epoch": 0.19186338523355098, "loss": 0.0735, "grad_norm": 7.080837726593018, "learning_rate": 4.215151515151515e-06, "num_tokens": 3485212.0, "completions/mean_length": 115.25, "completions/min_length": 99.0, "completions/max_length": 151.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.25, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 151.0, "rewards/meter/mean": 0.8732699751853943, "rewards/meter/std": 0.34030112624168396, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7260746955871582, "rewards/repeat_soft/std": 0.09545375406742096, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.1249857097864151, "rewards/total_composite/mean": 0.6932039856910706, "rewards/total_composite/std": 0.17277351021766663, "reward": 0.6932039856910706, "reward_std": 0.17277351021766663, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07849155366420746, "sampling/sampling_logp_difference/max": 1.593364953994751, "sampling/importance_sampling_ratio/min": 0.2032405585050583, "sampling/importance_sampling_ratio/mean": 1.0079907178878784, "sampling/importance_sampling_ratio/max": 1.7562129497528076, "entropy": 0.45538732036948204, "clip_ratio/low_mean": 0.012773487716913223, "clip_ratio/low_min": 0.012773487716913223, "clip_ratio/high_mean": 0.0661314488388598, "clip_ratio/high_max": 0.0661314488388598, "clip_ratio/region_mean": 0.07890493655577302, "reward_total_mean": 0.6932039856910706, "reward_meter_mean": 0.8732699751853943, "reward_meter_std": 0.34030112624168396, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7260746955871582, "reward_repeat_soft_std": 0.09545375406742096, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.1249857097864151, "reward_total_composite_mean": 0.6932039856910706, "reward_total_composite_std": 0.17277351021766663} {"timestamp_utc": "2026-04-13T03:08:00Z", "mode": "train", "global_step": 1911, "epoch": 0.19196383726770466, "loss": 0.0238, "grad_norm": 7.3372416496276855, "learning_rate": 4.212121212121212e-06, "num_tokens": 3487367.0, "completions/mean_length": 93.375, "completions/min_length": 89.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.375, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.950829267501831, "rewards/meter/std": 0.05468397215008736, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9240703582763672, "rewards/repeat_soft/std": 0.03700007498264313, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7899051904678345, "rewards/total_composite/std": 0.03732263296842575, "reward": 0.7899051904678345, "reward_std": 0.03732263296842575, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1001863107085228, "sampling/sampling_logp_difference/max": 1.510965347290039, "sampling/importance_sampling_ratio/min": 0.2947847843170166, "sampling/importance_sampling_ratio/mean": 1.0042355060577393, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5212728753685951, "clip_ratio/low_mean": 0.021341816056519747, "clip_ratio/low_min": 0.021341816056519747, "clip_ratio/high_mean": 0.0879699969664216, "clip_ratio/high_max": 0.0879699969664216, "clip_ratio/region_mean": 0.10931181302294135, "reward_total_mean": 0.7899051904678345, "reward_meter_mean": 0.950829267501831, "reward_meter_std": 0.05468397215008736, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9240703582763672, "reward_repeat_soft_std": 0.03700007498264313, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7899051904678345, "reward_total_composite_std": 0.03732263296842575} {"timestamp_utc": "2026-04-13T03:08:07Z", "mode": "train", "global_step": 1912, "epoch": 0.19206428930185837, "loss": 0.0336, "grad_norm": 5.664048194885254, "learning_rate": 4.2090909090909095e-06, "num_tokens": 3489845.0, "completions/mean_length": 122.75, "completions/min_length": 100.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.75, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.9890890121459961, "rewards/meter/std": 0.008990486152470112, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7812837958335876, "rewards/repeat_soft/std": 0.12591873109340668, "rewards/judge_quality/mean": 0.19625000655651093, "rewards/judge_quality/std": 0.09694438427686691, "rewards/total_composite/mean": 0.7058433890342712, "rewards/total_composite/std": 0.0274994857609272, "reward": 0.7058433890342712, "reward_std": 0.02749948389828205, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08384700864553452, "sampling/sampling_logp_difference/max": 1.8276453018188477, "sampling/importance_sampling_ratio/min": 0.16079173982143402, "sampling/importance_sampling_ratio/mean": 1.0066297054290771, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5012382566928864, "clip_ratio/low_mean": 0.03251290007028729, "clip_ratio/low_min": 0.03251290007028729, "clip_ratio/high_mean": 0.037931034341454506, "clip_ratio/high_max": 0.037931034341454506, "clip_ratio/region_mean": 0.0704439344117418, "reward_total_mean": 0.7058433890342712, "reward_meter_mean": 0.9890890121459961, "reward_meter_std": 0.008990486152470112, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7812837958335876, "reward_repeat_soft_std": 0.12591873109340668, "reward_judge_quality_mean": 0.19625000655651093, "reward_judge_quality_std": 0.09694438427686691, "reward_total_composite_mean": 0.7058433890342712, "reward_total_composite_std": 0.0274994857609272} {"timestamp_utc": "2026-04-13T03:08:14Z", "mode": "train", "global_step": 1913, "epoch": 0.19216474133601205, "loss": -0.0076, "grad_norm": 6.779642581939697, "learning_rate": 4.206060606060606e-06, "num_tokens": 3492292.0, "completions/mean_length": 125.875, "completions/min_length": 110.0, "completions/max_length": 165.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.875, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 165.0, "rewards/meter/mean": 0.9941658973693848, "rewards/meter/std": 0.0027343197725713253, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8985493183135986, "rewards/repeat_soft/std": 0.030354976654052734, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.12351980805397034, "rewards/total_composite/mean": 0.7474170923233032, "rewards/total_composite/std": 0.04577657952904701, "reward": 0.7474170923233032, "reward_std": 0.04577657952904701, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09528829902410507, "sampling/sampling_logp_difference/max": 3.4806530475616455, "sampling/importance_sampling_ratio/min": 0.03078729845583439, "sampling/importance_sampling_ratio/mean": 1.0034418106079102, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4535132758319378, "clip_ratio/low_mean": 0.04240168072283268, "clip_ratio/low_min": 0.04240168072283268, "clip_ratio/high_mean": 0.045301686972379684, "clip_ratio/high_max": 0.045301686972379684, "clip_ratio/region_mean": 0.08770336769521236, "reward_total_mean": 0.7474170923233032, "reward_meter_mean": 0.9941658973693848, "reward_meter_std": 0.0027343197725713253, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8985493183135986, "reward_repeat_soft_std": 0.030354976654052734, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.12351980805397034, "reward_total_composite_mean": 0.7474170923233032, "reward_total_composite_std": 0.04577657952904701} {"timestamp_utc": "2026-04-13T03:08:21Z", "mode": "train", "global_step": 1914, "epoch": 0.19226519337016573, "loss": 0.0542, "grad_norm": 8.984176635742188, "learning_rate": 4.203030303030303e-06, "num_tokens": 3494001.0, "completions/mean_length": 61.625, "completions/min_length": 53.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.625, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.7693608403205872, "rewards/meter/std": 0.3497885465621948, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9663844108581543, "rewards/repeat_soft/std": 0.021364498883485794, "rewards/judge_quality/mean": 0.42124998569488525, "rewards/judge_quality/std": 0.0699872374534607, "rewards/total_composite/mean": 0.7192258238792419, "rewards/total_composite/std": 0.15197688341140747, "reward": 0.7192258238792419, "reward_std": 0.15197686851024628, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0994749367237091, "sampling/sampling_logp_difference/max": 2.7011353969573975, "sampling/importance_sampling_ratio/min": 0.06712925434112549, "sampling/importance_sampling_ratio/mean": 1.0031025409698486, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31323040649294853, "clip_ratio/low_mean": 0.023324494250118732, "clip_ratio/low_min": 0.023324494250118732, "clip_ratio/high_mean": 0.06564056035131216, "clip_ratio/high_max": 0.06564056035131216, "clip_ratio/region_mean": 0.08896505460143089, "reward_total_mean": 0.7192258238792419, "reward_meter_mean": 0.7693608403205872, "reward_meter_std": 0.3497885465621948, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9663844108581543, "reward_repeat_soft_std": 0.021364498883485794, "reward_judge_quality_mean": 0.42124998569488525, "reward_judge_quality_std": 0.0699872374534607, "reward_total_composite_mean": 0.7192258238792419, "reward_total_composite_std": 0.15197688341140747} {"timestamp_utc": "2026-04-13T03:08:29Z", "mode": "train", "global_step": 1915, "epoch": 0.19236564540431944, "loss": 0.0447, "grad_norm": 4.183748722076416, "learning_rate": 4.2000000000000004e-06, "num_tokens": 3497107.0, "completions/mean_length": 191.25, "completions/min_length": 173.0, "completions/max_length": 205.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 191.25, "completions/min_terminated_length": 173.0, "completions/max_terminated_length": 205.0, "rewards/meter/mean": 0.9134717583656311, "rewards/meter/std": 0.13323155045509338, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7537325620651245, "rewards/repeat_soft/std": 0.05652475729584694, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.6564355492591858, "rewards/total_composite/std": 0.06011544540524483, "reward": 0.6564355492591858, "reward_std": 0.06011543795466423, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06512273848056793, "sampling/sampling_logp_difference/max": 3.676403045654297, "sampling/importance_sampling_ratio/min": 0.025313863530755043, "sampling/importance_sampling_ratio/mean": 0.9995765089988708, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2995439115911722, "clip_ratio/low_mean": 0.01691035646945238, "clip_ratio/low_min": 0.01691035646945238, "clip_ratio/high_mean": 0.046892013400793076, "clip_ratio/high_max": 0.046892013400793076, "clip_ratio/region_mean": 0.06380236987024546, "reward_total_mean": 0.6564355492591858, "reward_meter_mean": 0.9134717583656311, "reward_meter_std": 0.13323155045509338, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7537325620651245, "reward_repeat_soft_std": 0.05652475729584694, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.6564355492591858, "reward_total_composite_std": 0.06011544540524483} {"timestamp_utc": "2026-04-13T03:08:37Z", "mode": "train", "global_step": 1916, "epoch": 0.19246609743847312, "loss": 0.0146, "grad_norm": 7.095964431762695, "learning_rate": 4.196969696969697e-06, "num_tokens": 3499866.0, "completions/mean_length": 148.875, "completions/min_length": 125.0, "completions/max_length": 168.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 148.875, "completions/min_terminated_length": 125.0, "completions/max_terminated_length": 168.0, "rewards/meter/mean": 0.8570932745933533, "rewards/meter/std": 0.31208547949790955, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8272318840026855, "rewards/repeat_soft/std": 0.07843192666769028, "rewards/judge_quality/mean": 0.2549999952316284, "rewards/judge_quality/std": 0.1452092081308365, "rewards/total_composite/mean": 0.6855401396751404, "rewards/total_composite/std": 0.15303707122802734, "reward": 0.6855401396751404, "reward_std": 0.15303708612918854, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08239330351352692, "sampling/sampling_logp_difference/max": 2.010406017303467, "sampling/importance_sampling_ratio/min": 0.13393427431583405, "sampling/importance_sampling_ratio/mean": 1.0024168491363525, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40377791598439217, "clip_ratio/low_mean": 0.019778777845203876, "clip_ratio/low_min": 0.019778777845203876, "clip_ratio/high_mean": 0.055656439159065485, "clip_ratio/high_max": 0.055656439159065485, "clip_ratio/region_mean": 0.07543521700426936, "reward_total_mean": 0.6855401396751404, "reward_meter_mean": 0.8570932745933533, "reward_meter_std": 0.31208547949790955, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8272318840026855, "reward_repeat_soft_std": 0.07843192666769028, "reward_judge_quality_mean": 0.2549999952316284, "reward_judge_quality_std": 0.1452092081308365, "reward_total_composite_mean": 0.6855401396751404, "reward_total_composite_std": 0.15303707122802734} {"timestamp_utc": "2026-04-13T03:08:43Z", "mode": "train", "global_step": 1917, "epoch": 0.19256654947262683, "loss": 0.0642, "grad_norm": 11.488398551940918, "learning_rate": 4.193939393939394e-06, "num_tokens": 3501414.0, "completions/mean_length": 46.5, "completions/min_length": 42.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.5, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.7446509599685669, "rewards/meter/std": 0.3315240144729614, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.954669713973999, "rewards/repeat_soft/std": 0.047469571232795715, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.71930992603302, "rewards/total_composite/std": 0.1607864946126938, "reward": 0.71930992603302, "reward_std": 0.1607864946126938, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11693745106458664, "sampling/sampling_logp_difference/max": 1.755861759185791, "sampling/importance_sampling_ratio/min": 0.1727583110332489, "sampling/importance_sampling_ratio/mean": 1.0019770860671997, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4735201820731163, "clip_ratio/low_mean": 0.03182561625726521, "clip_ratio/low_min": 0.03182561625726521, "clip_ratio/high_mean": 0.0761335608549416, "clip_ratio/high_max": 0.0761335608549416, "clip_ratio/region_mean": 0.10795917711220682, "reward_total_mean": 0.71930992603302, "reward_meter_mean": 0.7446509599685669, "reward_meter_std": 0.3315240144729614, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.954669713973999, "reward_repeat_soft_std": 0.047469571232795715, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.71930992603302, "reward_total_composite_std": 0.1607864946126938} {"timestamp_utc": "2026-04-13T03:08:49Z", "mode": "train", "global_step": 1918, "epoch": 0.1926670015067805, "loss": 0.0438, "grad_norm": 14.623645782470703, "learning_rate": 4.190909090909091e-06, "num_tokens": 3502836.0, "completions/mean_length": 27.75, "completions/min_length": 25.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.75, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.9778212904930115, "rewards/meter/std": 0.018580932170152664, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8178945779800415, "rewards/total_composite/std": 0.010373925790190697, "reward": 0.8178945779800415, "reward_std": 0.010373924858868122, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08033433556556702, "sampling/sampling_logp_difference/max": 1.660961627960205, "sampling/importance_sampling_ratio/min": 0.18995621800422668, "sampling/importance_sampling_ratio/mean": 0.9950923323631287, "sampling/importance_sampling_ratio/max": 1.4874517917633057, "entropy": 0.40231676027178764, "clip_ratio/low_mean": 0.022268332540988922, "clip_ratio/low_min": 0.022268332540988922, "clip_ratio/high_mean": 0.04611111152917147, "clip_ratio/high_max": 0.04611111152917147, "clip_ratio/region_mean": 0.06837944407016039, "reward_total_mean": 0.8178945779800415, "reward_meter_mean": 0.9778212904930115, "reward_meter_std": 0.018580932170152664, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8178945779800415, "reward_total_composite_std": 0.010373925790190697} {"timestamp_utc": "2026-04-13T03:09:00Z", "mode": "train", "global_step": 1919, "epoch": 0.1927674535409342, "loss": -0.111, "grad_norm": 2.7152225971221924, "learning_rate": 4.187878787878788e-06, "num_tokens": 3504527.0, "completions/mean_length": 116.375, "completions/min_length": 54.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 59.857147216796875, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9656570553779602, "rewards/meter/std": 0.03506196290254593, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9555487632751465, "rewards/repeat_soft/std": 0.03889460116624832, "rewards/judge_quality/mean": 0.29999998211860657, "rewards/judge_quality/std": 0.13093072175979614, "rewards/total_composite/mean": 0.7607255578041077, "rewards/total_composite/std": 0.04976794868707657, "reward": 0.7607255578041077, "reward_std": 0.04976797103881836, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11412648856639862, "sampling/sampling_logp_difference/max": 1.6478188037872314, "sampling/importance_sampling_ratio/min": 0.1924692690372467, "sampling/importance_sampling_ratio/mean": 0.9908957481384277, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4407109245657921, "clip_ratio/low_mean": 0.024801588151603937, "clip_ratio/low_min": 0.024801588151603937, "clip_ratio/high_mean": 0.07189567061141133, "clip_ratio/high_max": 0.07189567061141133, "clip_ratio/region_mean": 0.09669725876301527, "reward_total_mean": 0.7607255578041077, "reward_meter_mean": 0.9656570553779602, "reward_meter_std": 0.03506196290254593, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9555487632751465, "reward_repeat_soft_std": 0.03889460116624832, "reward_judge_quality_mean": 0.29999998211860657, "reward_judge_quality_std": 0.13093072175979614, "reward_total_composite_mean": 0.7607255578041077, "reward_total_composite_std": 0.04976794868707657} {"timestamp_utc": "2026-04-13T03:09:08Z", "mode": "train", "global_step": 1920, "epoch": 0.1928679055750879, "loss": -0.0879, "grad_norm": 4.28872537612915, "learning_rate": 4.184848484848485e-06, "num_tokens": 3507232.0, "completions/mean_length": 132.125, "completions/min_length": 101.0, "completions/max_length": 166.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.125, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 166.0, "rewards/meter/mean": 0.9888288974761963, "rewards/meter/std": 0.0033041127026081085, "rewards/count_adherence/mean": 0.8500000238418579, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6081994771957397, "rewards/repeat_soft/std": 0.28153616189956665, "rewards/judge_quality/mean": 0.23000000417232513, "rewards/judge_quality/std": 0.12224097549915314, "rewards/total_composite/mean": 0.7022930383682251, "rewards/total_composite/std": 0.06060390919446945, "reward": 0.7022930383682251, "reward_std": 0.06060390919446945, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06209290027618408, "sampling/sampling_logp_difference/max": 1.78188157081604, "sampling/importance_sampling_ratio/min": 0.1683211475610733, "sampling/importance_sampling_ratio/mean": 1.0022004842758179, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31026701256632805, "clip_ratio/low_mean": 0.021247222903184593, "clip_ratio/low_min": 0.021247222903184593, "clip_ratio/high_mean": 0.04181759059429169, "clip_ratio/high_max": 0.04181759059429169, "clip_ratio/region_mean": 0.06306481349747628, "reward_total_mean": 0.7022930383682251, "reward_meter_mean": 0.9888288974761963, "reward_meter_std": 0.0033041127026081085, "reward_count_adherence_mean": 0.8500000238418579, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6081994771957397, "reward_repeat_soft_std": 0.28153616189956665, "reward_judge_quality_mean": 0.23000000417232513, "reward_judge_quality_std": 0.12224097549915314, "reward_total_composite_mean": 0.7022930383682251, "reward_total_composite_std": 0.06060390919446945} {"timestamp_utc": "2026-04-13T03:09:15Z", "mode": "train", "global_step": 1921, "epoch": 0.19296835760924158, "loss": 0.0556, "grad_norm": 11.993938446044922, "learning_rate": 4.181818181818182e-06, "num_tokens": 3508681.0, "completions/mean_length": 31.125, "completions/min_length": 26.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.125, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9207886457443237, "rewards/meter/std": 0.19064246118068695, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9490084648132324, "rewards/repeat_soft/std": 0.02298586256802082, "rewards/judge_quality/mean": 0.44624999165534973, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7931307554244995, "rewards/total_composite/std": 0.08483035862445831, "reward": 0.7931307554244995, "reward_std": 0.08483036607503891, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10138096660375595, "sampling/sampling_logp_difference/max": 1.3333330154418945, "sampling/importance_sampling_ratio/min": 0.32723525166511536, "sampling/importance_sampling_ratio/mean": 0.9961424469947815, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5849597826600075, "clip_ratio/low_mean": 0.02083333395421505, "clip_ratio/low_min": 0.02083333395421505, "clip_ratio/high_mean": 0.10497728642076254, "clip_ratio/high_max": 0.10497728642076254, "clip_ratio/region_mean": 0.1258106203749776, "reward_total_mean": 0.7931307554244995, "reward_meter_mean": 0.9207886457443237, "reward_meter_std": 0.19064246118068695, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9490084648132324, "reward_repeat_soft_std": 0.02298586256802082, "reward_judge_quality_mean": 0.44624999165534973, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7931307554244995, "reward_total_composite_std": 0.08483035862445831} {"timestamp_utc": "2026-04-13T03:09:21Z", "mode": "train", "global_step": 1922, "epoch": 0.1930688096433953, "loss": 0.0243, "grad_norm": 10.733656883239746, "learning_rate": 4.1787878787878795e-06, "num_tokens": 3510481.0, "completions/mean_length": 62.0, "completions/min_length": 60.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.6572411060333252, "rewards/meter/std": 0.3669791519641876, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9539139270782471, "rewards/repeat_soft/std": 0.028199080377817154, "rewards/judge_quality/mean": 0.39625000953674316, "rewards/judge_quality/std": 0.09085899591445923, "rewards/total_composite/mean": 0.660024881362915, "rewards/total_composite/std": 0.17496956884860992, "reward": 0.660024881362915, "reward_std": 0.17496956884860992, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12695477902889252, "sampling/sampling_logp_difference/max": 1.5788171291351318, "sampling/importance_sampling_ratio/min": 0.206218883395195, "sampling/importance_sampling_ratio/mean": 0.9969602823257446, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7263700366020203, "clip_ratio/low_mean": 0.03970114188268781, "clip_ratio/low_min": 0.03970114188268781, "clip_ratio/high_mean": 0.08729508332908154, "clip_ratio/high_max": 0.08729508332908154, "clip_ratio/region_mean": 0.12699622521176934, "reward_total_mean": 0.660024881362915, "reward_meter_mean": 0.6572411060333252, "reward_meter_std": 0.3669791519641876, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9539139270782471, "reward_repeat_soft_std": 0.028199080377817154, "reward_judge_quality_mean": 0.39625000953674316, "reward_judge_quality_std": 0.09085899591445923, "reward_total_composite_mean": 0.660024881362915, "reward_total_composite_std": 0.17496956884860992} {"timestamp_utc": "2026-04-13T03:09:30Z", "mode": "train", "global_step": 1923, "epoch": 0.19316926167754897, "loss": 0.028, "grad_norm": 9.520705223083496, "learning_rate": 4.175757575757576e-06, "num_tokens": 3512296.0, "completions/mean_length": 70.875, "completions/min_length": 60.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.875, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.7703976035118103, "rewards/meter/std": 0.2603209614753723, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9332410097122192, "rewards/repeat_soft/std": 0.05451788753271103, "rewards/judge_quality/mean": 0.41749998927116394, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.7152529954910278, "rewards/total_composite/std": 0.12825718522071838, "reward": 0.7152529954910278, "reward_std": 0.12825718522071838, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1172475591301918, "sampling/sampling_logp_difference/max": 1.7683382034301758, "sampling/importance_sampling_ratio/min": 0.1706162840127945, "sampling/importance_sampling_ratio/mean": 0.9910747408866882, "sampling/importance_sampling_ratio/max": 1.9679107666015625, "entropy": 0.6036684438586235, "clip_ratio/low_mean": 0.02724156156182289, "clip_ratio/low_min": 0.02724156156182289, "clip_ratio/high_mean": 0.09485246147960424, "clip_ratio/high_max": 0.09485246147960424, "clip_ratio/region_mean": 0.12209402304142714, "reward_total_mean": 0.7152529954910278, "reward_meter_mean": 0.7703976035118103, "reward_meter_std": 0.2603209614753723, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9332410097122192, "reward_repeat_soft_std": 0.05451788753271103, "reward_judge_quality_mean": 0.41749998927116394, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.7152529954910278, "reward_total_composite_std": 0.12825718522071838} {"timestamp_utc": "2026-04-13T03:09:36Z", "mode": "train", "global_step": 1924, "epoch": 0.19326971371170265, "loss": 0.0229, "grad_norm": 9.391199111938477, "learning_rate": 4.172727272727273e-06, "num_tokens": 3513934.0, "completions/mean_length": 46.75, "completions/min_length": 39.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.75, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.7656022310256958, "rewards/meter/std": 0.27133703231811523, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9305135011672974, "rewards/repeat_soft/std": 0.04570186510682106, "rewards/judge_quality/mean": 0.3762499988079071, "rewards/judge_quality/std": 0.11287634819746017, "rewards/total_composite/mean": 0.7004473209381104, "rewards/total_composite/std": 0.13547150790691376, "reward": 0.7004473209381104, "reward_std": 0.13547149300575256, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09774432331323624, "sampling/sampling_logp_difference/max": 2.0641307830810547, "sampling/importance_sampling_ratio/min": 0.12692856788635254, "sampling/importance_sampling_ratio/mean": 1.01937997341156, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49084774777293205, "clip_ratio/low_mean": 0.02099955640733242, "clip_ratio/low_min": 0.02099955640733242, "clip_ratio/high_mean": 0.04927911749109626, "clip_ratio/high_max": 0.04927911749109626, "clip_ratio/region_mean": 0.07027867389842868, "reward_total_mean": 0.7004473209381104, "reward_meter_mean": 0.7656022310256958, "reward_meter_std": 0.27133703231811523, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9305135011672974, "reward_repeat_soft_std": 0.04570186510682106, "reward_judge_quality_mean": 0.3762499988079071, "reward_judge_quality_std": 0.11287634819746017, "reward_total_composite_mean": 0.7004473209381104, "reward_total_composite_std": 0.13547150790691376} {"timestamp_utc": "2026-04-13T03:09:43Z", "mode": "train", "global_step": 1925, "epoch": 0.19337016574585636, "loss": -0.0293, "grad_norm": 9.300021171569824, "learning_rate": 4.1696969696969705e-06, "num_tokens": 3515585.0, "completions/mean_length": 60.375, "completions/min_length": 54.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.375, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9109476208686829, "rewards/meter/std": 0.19678516685962677, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.921906590461731, "rewards/repeat_soft/std": 0.0889788568019867, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.7762420773506165, "rewards/total_composite/std": 0.0943308100104332, "reward": 0.7762420773506165, "reward_std": 0.0943308100104332, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10986492782831192, "sampling/sampling_logp_difference/max": 2.1184256076812744, "sampling/importance_sampling_ratio/min": 0.12022075802087784, "sampling/importance_sampling_ratio/mean": 1.0112290382385254, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6413133889436722, "clip_ratio/low_mean": 0.01330037647858262, "clip_ratio/low_min": 0.01330037647858262, "clip_ratio/high_mean": 0.0858142925426364, "clip_ratio/high_max": 0.0858142925426364, "clip_ratio/region_mean": 0.09911466902121902, "reward_total_mean": 0.7762420773506165, "reward_meter_mean": 0.9109476208686829, "reward_meter_std": 0.19678516685962677, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.921906590461731, "reward_repeat_soft_std": 0.0889788568019867, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.7762420773506165, "reward_total_composite_std": 0.0943308100104332} {"timestamp_utc": "2026-04-13T03:09:51Z", "mode": "train", "global_step": 1926, "epoch": 0.19347061778001004, "loss": -0.0007, "grad_norm": 7.305087089538574, "learning_rate": 4.166666666666667e-06, "num_tokens": 3517546.0, "completions/mean_length": 74.125, "completions/min_length": 68.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.125, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.9915218353271484, "rewards/meter/std": 0.0019607648719102144, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9001253843307495, "rewards/repeat_soft/std": 0.05657697841525078, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8121973276138306, "rewards/total_composite/std": 0.005498938262462616, "reward": 0.8121973276138306, "reward_std": 0.005498944316059351, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07419052720069885, "sampling/sampling_logp_difference/max": 1.9901390075683594, "sampling/importance_sampling_ratio/min": 0.13667643070220947, "sampling/importance_sampling_ratio/mean": 1.0016865730285645, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3714553862810135, "clip_ratio/low_mean": 0.0170932007022202, "clip_ratio/low_min": 0.0170932007022202, "clip_ratio/high_mean": 0.04103752668015659, "clip_ratio/high_max": 0.04103752668015659, "clip_ratio/region_mean": 0.05813072738237679, "reward_total_mean": 0.8121973276138306, "reward_meter_mean": 0.9915218353271484, "reward_meter_std": 0.0019607648719102144, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9001253843307495, "reward_repeat_soft_std": 0.05657697841525078, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8121973276138306, "reward_total_composite_std": 0.005498938262462616} {"timestamp_utc": "2026-04-13T03:09:57Z", "mode": "train", "global_step": 1927, "epoch": 0.19357106981416375, "loss": -0.0396, "grad_norm": 13.647696495056152, "learning_rate": 4.163636363636364e-06, "num_tokens": 3519110.0, "completions/mean_length": 36.5, "completions/min_length": 26.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.5, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9862115979194641, "rewards/meter/std": 0.0074724191799759865, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9460016489028931, "rewards/repeat_soft/std": 0.023951048031449318, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.820020318031311, "rewards/total_composite/std": 0.006053817458450794, "reward": 0.820020318031311, "reward_std": 0.006053818389773369, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10996229201555252, "sampling/sampling_logp_difference/max": 0.9716792106628418, "sampling/importance_sampling_ratio/min": 0.378447026014328, "sampling/importance_sampling_ratio/mean": 0.9977943897247314, "sampling/importance_sampling_ratio/max": 1.789272427558899, "entropy": 0.601938109844923, "clip_ratio/low_mean": 0.04830586211755872, "clip_ratio/low_min": 0.04830586211755872, "clip_ratio/high_mean": 0.09452374465763569, "clip_ratio/high_max": 0.09452374465763569, "clip_ratio/region_mean": 0.1428296067751944, "reward_total_mean": 0.820020318031311, "reward_meter_mean": 0.9862115979194641, "reward_meter_std": 0.0074724191799759865, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9460016489028931, "reward_repeat_soft_std": 0.023951048031449318, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.820020318031311, "reward_total_composite_std": 0.006053817458450794} {"timestamp_utc": "2026-04-13T03:10:04Z", "mode": "train", "global_step": 1928, "epoch": 0.19367152184831743, "loss": 0.0991, "grad_norm": 14.702091217041016, "learning_rate": 4.160606060606061e-06, "num_tokens": 3520533.0, "completions/mean_length": 32.875, "completions/min_length": 28.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.875, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.7721225023269653, "rewards/meter/std": 0.3433400094509125, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9510133266448975, "rewards/repeat_soft/std": 0.0245317704975605, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.19255799055099487, "rewards/total_composite/mean": 0.7343064546585083, "rewards/total_composite/std": 0.15981172025203705, "reward": 0.7343064546585083, "reward_std": 0.15981170535087585, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09931809455156326, "sampling/sampling_logp_difference/max": 1.2448701858520508, "sampling/importance_sampling_ratio/min": 0.28797829151153564, "sampling/importance_sampling_ratio/mean": 0.9834211468696594, "sampling/importance_sampling_ratio/max": 1.5682895183563232, "entropy": 0.636726375669241, "clip_ratio/low_mean": 0.03440799564123154, "clip_ratio/low_min": 0.03440799564123154, "clip_ratio/high_mean": 0.09231601981446147, "clip_ratio/high_max": 0.09231601981446147, "clip_ratio/region_mean": 0.126724015455693, "reward_total_mean": 0.7343064546585083, "reward_meter_mean": 0.7721225023269653, "reward_meter_std": 0.3433400094509125, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9510133266448975, "reward_repeat_soft_std": 0.0245317704975605, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.19255799055099487, "reward_total_composite_mean": 0.7343064546585083, "reward_total_composite_std": 0.15981172025203705} {"timestamp_utc": "2026-04-13T03:10:15Z", "mode": "train", "global_step": 1929, "epoch": 0.1937719738824711, "loss": -0.1047, "grad_norm": 1.1956710815429688, "learning_rate": 4.157575757575758e-06, "num_tokens": 3522151.0, "completions/mean_length": 176.25, "completions/min_length": 57.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 64.33333587646484, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.8494611382484436, "rewards/meter/std": 0.3444003164768219, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9144511222839355, "rewards/repeat_soft/std": 0.11328868567943573, "rewards/judge_quality/mean": 0.29749998450279236, "rewards/judge_quality/std": 0.14518460631370544, "rewards/total_composite/mean": 0.6798276305198669, "rewards/total_composite/std": 0.27880728244781494, "reward": 0.6798276305198669, "reward_std": 0.27880731225013733, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09351030737161636, "sampling/sampling_logp_difference/max": 1.1655516624450684, "sampling/importance_sampling_ratio/min": 0.3117506504058838, "sampling/importance_sampling_ratio/mean": 1.0105066299438477, "sampling/importance_sampling_ratio/max": 1.8483169078826904, "entropy": 0.4129173010587692, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.07201811578124762, "clip_ratio/high_max": 0.07201811578124762, "clip_ratio/region_mean": 0.07201811578124762, "reward_total_mean": 0.6798276305198669, "reward_meter_mean": 0.8494611382484436, "reward_meter_std": 0.3444003164768219, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9144511222839355, "reward_repeat_soft_std": 0.11328868567943573, "reward_judge_quality_mean": 0.29749998450279236, "reward_judge_quality_std": 0.14518460631370544, "reward_total_composite_mean": 0.6798276305198669, "reward_total_composite_std": 0.27880728244781494} {"timestamp_utc": "2026-04-13T03:10:22Z", "mode": "train", "global_step": 1930, "epoch": 0.19387242591662482, "loss": -0.0237, "grad_norm": 9.489243507385254, "learning_rate": 4.154545454545455e-06, "num_tokens": 3524040.0, "completions/mean_length": 77.125, "completions/min_length": 70.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.125, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.8692787885665894, "rewards/meter/std": 0.27206337451934814, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9606850147247314, "rewards/repeat_soft/std": 0.026423379778862, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7441189289093018, "rewards/total_composite/std": 0.11490379273891449, "reward": 0.7441189289093018, "reward_std": 0.1149037778377533, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10157779604196548, "sampling/sampling_logp_difference/max": 1.447103500366211, "sampling/importance_sampling_ratio/min": 0.23525071144104004, "sampling/importance_sampling_ratio/mean": 1.007201910018921, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5729658454656601, "clip_ratio/low_mean": 0.008928571827709675, "clip_ratio/low_min": 0.008928571827709675, "clip_ratio/high_mean": 0.08466806821525097, "clip_ratio/high_max": 0.08466806821525097, "clip_ratio/region_mean": 0.09359664004296064, "reward_total_mean": 0.7441189289093018, "reward_meter_mean": 0.8692787885665894, "reward_meter_std": 0.27206337451934814, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9606850147247314, "reward_repeat_soft_std": 0.026423379778862, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7441189289093018, "reward_total_composite_std": 0.11490379273891449} {"timestamp_utc": "2026-04-13T03:10:30Z", "mode": "train", "global_step": 1931, "epoch": 0.1939728779507785, "loss": 0.0301, "grad_norm": 5.020719528198242, "learning_rate": 4.151515151515152e-06, "num_tokens": 3525831.0, "completions/mean_length": 74.875, "completions/min_length": 64.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.875, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.8701591491699219, "rewards/meter/std": 0.30188971757888794, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8924897313117981, "rewards/repeat_soft/std": 0.06091193109750748, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.7755706310272217, "rewards/total_composite/std": 0.15186063945293427, "reward": 0.7755706310272217, "reward_std": 0.15186063945293427, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08949016779661179, "sampling/sampling_logp_difference/max": 1.6106460094451904, "sampling/importance_sampling_ratio/min": 0.19975851476192474, "sampling/importance_sampling_ratio/mean": 1.0019792318344116, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.42957474663853645, "clip_ratio/low_mean": 0.011075949296355247, "clip_ratio/low_min": 0.011075949296355247, "clip_ratio/high_mean": 0.09125571884214878, "clip_ratio/high_max": 0.09125571884214878, "clip_ratio/region_mean": 0.10233166813850403, "reward_total_mean": 0.7755706310272217, "reward_meter_mean": 0.8701591491699219, "reward_meter_std": 0.30188971757888794, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8924897313117981, "reward_repeat_soft_std": 0.06091193109750748, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.7755706310272217, "reward_total_composite_std": 0.15186063945293427} {"timestamp_utc": "2026-04-13T03:10:38Z", "mode": "train", "global_step": 1932, "epoch": 0.1940733299849322, "loss": -0.0096, "grad_norm": 6.726437091827393, "learning_rate": 4.148484848484849e-06, "num_tokens": 3528410.0, "completions/mean_length": 143.375, "completions/min_length": 122.0, "completions/max_length": 157.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 143.375, "completions/min_terminated_length": 122.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.984881579875946, "rewards/meter/std": 0.007367500104010105, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8728092312812805, "rewards/repeat_soft/std": 0.054947808384895325, "rewards/judge_quality/mean": 0.19625000655651093, "rewards/judge_quality/std": 0.09694438427686691, "rewards/total_composite/mean": 0.7393526434898376, "rewards/total_composite/std": 0.03349566459655762, "reward": 0.7393526434898376, "reward_std": 0.03349565714597702, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08153142780065536, "sampling/sampling_logp_difference/max": 1.515209674835205, "sampling/importance_sampling_ratio/min": 0.21976211667060852, "sampling/importance_sampling_ratio/mean": 1.0106772184371948, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.473365131765604, "clip_ratio/low_mean": 0.06027217488735914, "clip_ratio/low_min": 0.06027217488735914, "clip_ratio/high_mean": 0.014751112554222345, "clip_ratio/high_max": 0.014751112554222345, "clip_ratio/region_mean": 0.07502328744158149, "reward_total_mean": 0.7393526434898376, "reward_meter_mean": 0.984881579875946, "reward_meter_std": 0.007367500104010105, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8728092312812805, "reward_repeat_soft_std": 0.054947808384895325, "reward_judge_quality_mean": 0.19625000655651093, "reward_judge_quality_std": 0.09694438427686691, "reward_total_composite_mean": 0.7393526434898376, "reward_total_composite_std": 0.03349566459655762} {"timestamp_utc": "2026-04-13T03:10:46Z", "mode": "train", "global_step": 1933, "epoch": 0.1941737820190859, "loss": -0.0029, "grad_norm": 9.287967681884766, "learning_rate": 4.145454545454546e-06, "num_tokens": 3531219.0, "completions/mean_length": 173.125, "completions/min_length": 154.0, "completions/max_length": 193.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 173.125, "completions/min_terminated_length": 154.0, "completions/max_terminated_length": 193.0, "rewards/meter/mean": 0.8742072582244873, "rewards/meter/std": 0.1665426790714264, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.14880475401878357, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8222953677177429, "rewards/repeat_soft/std": 0.04738221690058708, "rewards/judge_quality/mean": 0.15625, "rewards/judge_quality/std": 0.04172614961862564, "rewards/total_composite/mean": 0.6612477898597717, "rewards/total_composite/std": 0.09479645639657974, "reward": 0.6612477898597717, "reward_std": 0.09479645639657974, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08060058206319809, "sampling/sampling_logp_difference/max": 2.065521001815796, "sampling/importance_sampling_ratio/min": 0.12675222754478455, "sampling/importance_sampling_ratio/mean": 1.0014556646347046, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4044397361576557, "clip_ratio/low_mean": 0.011093073524534702, "clip_ratio/low_min": 0.011093073524534702, "clip_ratio/high_mean": 0.05683378363028169, "clip_ratio/high_max": 0.05683378363028169, "clip_ratio/region_mean": 0.06792685715481639, "reward_total_mean": 0.6612477898597717, "reward_meter_mean": 0.8742072582244873, "reward_meter_std": 0.1665426790714264, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.14880475401878357, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8222953677177429, "reward_repeat_soft_std": 0.04738221690058708, "reward_judge_quality_mean": 0.15625, "reward_judge_quality_std": 0.04172614961862564, "reward_total_composite_mean": 0.6612477898597717, "reward_total_composite_std": 0.09479645639657974} {"timestamp_utc": "2026-04-13T03:10:53Z", "mode": "train", "global_step": 1934, "epoch": 0.19427423405323957, "loss": 0.0017, "grad_norm": 8.470044136047363, "learning_rate": 4.142424242424243e-06, "num_tokens": 3533291.0, "completions/mean_length": 96.0, "completions/min_length": 84.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.0, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.9573773145675659, "rewards/meter/std": 0.035061489790678024, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9153150916099548, "rewards/repeat_soft/std": 0.05316166207194328, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7856013178825378, "rewards/total_composite/std": 0.025828387588262558, "reward": 0.7856013178825378, "reward_std": 0.025828387588262558, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09154751151800156, "sampling/sampling_logp_difference/max": 2.200307846069336, "sampling/importance_sampling_ratio/min": 0.11076904833316803, "sampling/importance_sampling_ratio/mean": 1.0023682117462158, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5097203440964222, "clip_ratio/low_mean": 0.03322981344535947, "clip_ratio/low_min": 0.03322981344535947, "clip_ratio/high_mean": 0.06828074622899294, "clip_ratio/high_max": 0.06828074622899294, "clip_ratio/region_mean": 0.10151055967435241, "reward_total_mean": 0.7856013178825378, "reward_meter_mean": 0.9573773145675659, "reward_meter_std": 0.035061489790678024, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9153150916099548, "reward_repeat_soft_std": 0.05316166207194328, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7856013178825378, "reward_total_composite_std": 0.025828387588262558} {"timestamp_utc": "2026-04-13T03:11:01Z", "mode": "train", "global_step": 1935, "epoch": 0.19437468608739328, "loss": 0.0262, "grad_norm": 6.344499111175537, "learning_rate": 4.13939393939394e-06, "num_tokens": 3535262.0, "completions/mean_length": 96.375, "completions/min_length": 90.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.375, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.6473625898361206, "rewards/meter/std": 0.34315043687820435, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9235497713088989, "rewards/repeat_soft/std": 0.05675523728132248, "rewards/judge_quality/mean": 0.24250000715255737, "rewards/judge_quality/std": 0.11792854964733124, "rewards/total_composite/mean": 0.6064181327819824, "rewards/total_composite/std": 0.17021457850933075, "reward": 0.6064181327819824, "reward_std": 0.17021456360816956, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09826597571372986, "sampling/sampling_logp_difference/max": 2.33950138092041, "sampling/importance_sampling_ratio/min": 0.09637568145990372, "sampling/importance_sampling_ratio/mean": 1.001368761062622, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4914188049733639, "clip_ratio/low_mean": 0.03628165530972183, "clip_ratio/low_min": 0.03628165530972183, "clip_ratio/high_mean": 0.043777670711278915, "clip_ratio/high_max": 0.043777670711278915, "clip_ratio/region_mean": 0.08005932602100074, "reward_total_mean": 0.6064181327819824, "reward_meter_mean": 0.6473625898361206, "reward_meter_std": 0.34315043687820435, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9235497713088989, "reward_repeat_soft_std": 0.05675523728132248, "reward_judge_quality_mean": 0.24250000715255737, "reward_judge_quality_std": 0.11792854964733124, "reward_total_composite_mean": 0.6064181327819824, "reward_total_composite_std": 0.17021457850933075} {"timestamp_utc": "2026-04-13T03:11:08Z", "mode": "train", "global_step": 1936, "epoch": 0.19447513812154696, "loss": 0.012, "grad_norm": 9.712798118591309, "learning_rate": 4.136363636363637e-06, "num_tokens": 3536991.0, "completions/mean_length": 59.125, "completions/min_length": 53.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.125, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9719897508621216, "rewards/meter/std": 0.03227829933166504, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9473720788955688, "rewards/repeat_soft/std": 0.05108179897069931, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720350325107574, "rewards/total_composite/mean": 0.8478826284408569, "rewards/total_composite/std": 0.07651933282613754, "reward": 0.8478826284408569, "reward_std": 0.07651934027671814, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09049911797046661, "sampling/sampling_logp_difference/max": 1.2581977844238281, "sampling/importance_sampling_ratio/min": 0.2841656804084778, "sampling/importance_sampling_ratio/mean": 1.0112930536270142, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5221898257732391, "clip_ratio/low_mean": 0.06930598383769393, "clip_ratio/low_min": 0.06930598383769393, "clip_ratio/high_mean": 0.02119252923876047, "clip_ratio/high_max": 0.02119252923876047, "clip_ratio/region_mean": 0.0904985130764544, "reward_total_mean": 0.8478826284408569, "reward_meter_mean": 0.9719897508621216, "reward_meter_std": 0.03227829933166504, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9473720788955688, "reward_repeat_soft_std": 0.05108179897069931, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720350325107574, "reward_total_composite_mean": 0.8478826284408569, "reward_total_composite_std": 0.07651933282613754} {"timestamp_utc": "2026-04-13T03:11:20Z", "mode": "train", "global_step": 1937, "epoch": 0.19457559015570064, "loss": -0.1551, "grad_norm": 1.86814284324646, "learning_rate": 4.133333333333333e-06, "num_tokens": 3538661.0, "completions/mean_length": 116.75, "completions/min_length": 54.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 60.28571701049805, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.8682789206504822, "rewards/meter/std": 0.32620397210121155, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9663939476013184, "rewards/repeat_soft/std": 0.03707921877503395, "rewards/judge_quality/mean": 0.39750000834465027, "rewards/judge_quality/std": 0.10110107809305191, "rewards/total_composite/mean": 0.747239887714386, "rewards/total_composite/std": 0.20364902913570404, "reward": 0.747239887714386, "reward_std": 0.20364901423454285, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12497600167989731, "sampling/sampling_logp_difference/max": 1.5587934255599976, "sampling/importance_sampling_ratio/min": 0.21038977801799774, "sampling/importance_sampling_ratio/mean": 1.0245096683502197, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5620756223797798, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10782963503152132, "clip_ratio/high_max": 0.10782963503152132, "clip_ratio/region_mean": 0.10782963503152132, "reward_total_mean": 0.747239887714386, "reward_meter_mean": 0.8682789206504822, "reward_meter_std": 0.32620397210121155, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9663939476013184, "reward_repeat_soft_std": 0.03707921877503395, "reward_judge_quality_mean": 0.39750000834465027, "reward_judge_quality_std": 0.10110107809305191, "reward_total_composite_mean": 0.747239887714386, "reward_total_composite_std": 0.20364902913570404} {"timestamp_utc": "2026-04-13T03:11:27Z", "mode": "train", "global_step": 1938, "epoch": 0.19467604218985435, "loss": 0.0989, "grad_norm": 15.618401527404785, "learning_rate": 4.1303030303030305e-06, "num_tokens": 3540117.0, "completions/mean_length": 37.0, "completions/min_length": 31.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.616729199886322, "rewards/meter/std": 0.46250590682029724, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9597537517547607, "rewards/repeat_soft/std": 0.005922909360378981, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.6517534852027893, "rewards/total_composite/std": 0.20983745157718658, "reward": 0.6517534852027893, "reward_std": 0.20983746647834778, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14068663120269775, "sampling/sampling_logp_difference/max": 1.846963882446289, "sampling/importance_sampling_ratio/min": 0.15771529078483582, "sampling/importance_sampling_ratio/mean": 0.9975414872169495, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7489631548523903, "clip_ratio/low_mean": 0.050101215951144695, "clip_ratio/low_min": 0.050101215951144695, "clip_ratio/high_mean": 0.12161821685731411, "clip_ratio/high_max": 0.12161821685731411, "clip_ratio/region_mean": 0.1717194328084588, "reward_total_mean": 0.6517534852027893, "reward_meter_mean": 0.616729199886322, "reward_meter_std": 0.46250590682029724, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9597537517547607, "reward_repeat_soft_std": 0.005922909360378981, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.6517534852027893, "reward_total_composite_std": 0.20983745157718658} {"timestamp_utc": "2026-04-13T03:11:37Z", "mode": "train", "global_step": 1939, "epoch": 0.19477649422400803, "loss": 0.0062, "grad_norm": 13.071807861328125, "learning_rate": 4.127272727272728e-06, "num_tokens": 3541653.0, "completions/mean_length": 37.0, "completions/min_length": 33.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.9844130277633667, "rewards/meter/std": 0.01107175461947918, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7750853300094604, "rewards/repeat_soft/std": 0.07572069764137268, "rewards/judge_quality/mean": 0.33375000953674316, "rewards/judge_quality/std": 0.1524970680475235, "rewards/total_composite/mean": 0.7706193923950195, "rewards/total_composite/std": 0.04826018586754799, "reward": 0.7706193923950195, "reward_std": 0.048260170966386795, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11401914060115814, "sampling/sampling_logp_difference/max": 1.3956880569458008, "sampling/importance_sampling_ratio/min": 0.24766258895397186, "sampling/importance_sampling_ratio/mean": 0.9812849164009094, "sampling/importance_sampling_ratio/max": 1.997760534286499, "entropy": 0.5174011774361134, "clip_ratio/low_mean": 0.03200843371450901, "clip_ratio/low_min": 0.03200843371450901, "clip_ratio/high_mean": 0.059709781780838966, "clip_ratio/high_max": 0.059709781780838966, "clip_ratio/region_mean": 0.09171821549534798, "reward_total_mean": 0.7706193923950195, "reward_meter_mean": 0.9844130277633667, "reward_meter_std": 0.01107175461947918, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7750853300094604, "reward_repeat_soft_std": 0.07572069764137268, "reward_judge_quality_mean": 0.33375000953674316, "reward_judge_quality_std": 0.1524970680475235, "reward_total_composite_mean": 0.7706193923950195, "reward_total_composite_std": 0.04826018586754799} {"timestamp_utc": "2026-04-13T03:11:44Z", "mode": "train", "global_step": 1940, "epoch": 0.19487694625816174, "loss": 0.0008, "grad_norm": 8.579609870910645, "learning_rate": 4.124242424242424e-06, "num_tokens": 3543727.0, "completions/mean_length": 81.25, "completions/min_length": 64.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.25, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.9929416179656982, "rewards/meter/std": 0.0038336673751473427, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8765814304351807, "rewards/repeat_soft/std": 0.04391656816005707, "rewards/judge_quality/mean": 0.23374998569488525, "rewards/judge_quality/std": 0.09006941318511963, "rewards/total_composite/mean": 0.7452319264411926, "rewards/total_composite/std": 0.03979345038533211, "reward": 0.7452319264411926, "reward_std": 0.03979344666004181, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08023151010274887, "sampling/sampling_logp_difference/max": 1.9362974166870117, "sampling/importance_sampling_ratio/min": 0.14423701167106628, "sampling/importance_sampling_ratio/mean": 1.013567328453064, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3871123045682907, "clip_ratio/low_mean": 0.036014047684147954, "clip_ratio/low_min": 0.036014047684147954, "clip_ratio/high_mean": 0.03658712701871991, "clip_ratio/high_max": 0.03658712701871991, "clip_ratio/region_mean": 0.07260117470286787, "reward_total_mean": 0.7452319264411926, "reward_meter_mean": 0.9929416179656982, "reward_meter_std": 0.0038336673751473427, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8765814304351807, "reward_repeat_soft_std": 0.04391656816005707, "reward_judge_quality_mean": 0.23374998569488525, "reward_judge_quality_std": 0.09006941318511963, "reward_total_composite_mean": 0.7452319264411926, "reward_total_composite_std": 0.03979345038533211} {"timestamp_utc": "2026-04-13T03:11:52Z", "mode": "train", "global_step": 1941, "epoch": 0.19497739829231542, "loss": 0.0928, "grad_norm": 9.01904582977295, "learning_rate": 4.1212121212121215e-06, "num_tokens": 3545715.0, "completions/mean_length": 72.5, "completions/min_length": 64.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.5, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.7653558850288391, "rewards/meter/std": 0.342529296875, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9582508206367493, "rewards/repeat_soft/std": 0.057376790791749954, "rewards/judge_quality/mean": 0.21250000596046448, "rewards/judge_quality/std": 0.0517549142241478, "rewards/total_composite/mean": 0.6539852619171143, "rewards/total_composite/std": 0.1593034714460373, "reward": 0.6539852619171143, "reward_std": 0.1593034714460373, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1165294274687767, "sampling/sampling_logp_difference/max": 1.3967349529266357, "sampling/importance_sampling_ratio/min": 0.24740344285964966, "sampling/importance_sampling_ratio/mean": 0.9861608147621155, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.590282529592514, "clip_ratio/low_mean": 0.01925545558333397, "clip_ratio/low_min": 0.01925545558333397, "clip_ratio/high_mean": 0.08403028082102537, "clip_ratio/high_max": 0.08403028082102537, "clip_ratio/region_mean": 0.10328573640435934, "reward_total_mean": 0.6539852619171143, "reward_meter_mean": 0.7653558850288391, "reward_meter_std": 0.342529296875, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9582508206367493, "reward_repeat_soft_std": 0.057376790791749954, "reward_judge_quality_mean": 0.21250000596046448, "reward_judge_quality_std": 0.0517549142241478, "reward_total_composite_mean": 0.6539852619171143, "reward_total_composite_std": 0.1593034714460373} {"timestamp_utc": "2026-04-13T03:11:59Z", "mode": "train", "global_step": 1942, "epoch": 0.1950778503264691, "loss": -0.0179, "grad_norm": 14.30864429473877, "learning_rate": 4.118181818181819e-06, "num_tokens": 3547368.0, "completions/mean_length": 32.625, "completions/min_length": 31.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.625, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.735015869140625, "rewards/meter/std": 0.42443618178367615, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9063470363616943, "rewards/repeat_soft/std": 0.04582662507891655, "rewards/judge_quality/mean": 0.26250001788139343, "rewards/judge_quality/std": 0.155264750123024, "rewards/total_composite/mean": 0.6501418352127075, "rewards/total_composite/std": 0.191147118806839, "reward": 0.6501418352127075, "reward_std": 0.191147118806839, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14948923885822296, "sampling/sampling_logp_difference/max": 1.4535207748413086, "sampling/importance_sampling_ratio/min": 0.23805022239685059, "sampling/importance_sampling_ratio/mean": 0.9807451367378235, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.703533485531807, "clip_ratio/low_mean": 0.024193547666072845, "clip_ratio/low_min": 0.024193547666072845, "clip_ratio/high_mean": 0.10530732944607735, "clip_ratio/high_max": 0.10530732944607735, "clip_ratio/region_mean": 0.1295008771121502, "reward_total_mean": 0.6501418352127075, "reward_meter_mean": 0.735015869140625, "reward_meter_std": 0.42443618178367615, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9063470363616943, "reward_repeat_soft_std": 0.04582662507891655, "reward_judge_quality_mean": 0.26250001788139343, "reward_judge_quality_std": 0.155264750123024, "reward_total_composite_mean": 0.6501418352127075, "reward_total_composite_std": 0.191147118806839} {"timestamp_utc": "2026-04-13T03:12:07Z", "mode": "train", "global_step": 1943, "epoch": 0.1951783023606228, "loss": -0.0942, "grad_norm": 7.902928829193115, "learning_rate": 4.115151515151515e-06, "num_tokens": 3549312.0, "completions/mean_length": 89.0, "completions/min_length": 70.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.0, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.9817661046981812, "rewards/meter/std": 0.014558358117938042, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.17251639068126678, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8345391750335693, "rewards/repeat_soft/std": 0.057019129395484924, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.1249857097864151, "rewards/total_composite/mean": 0.7622486352920532, "rewards/total_composite/std": 0.06372414529323578, "reward": 0.7622486352920532, "reward_std": 0.06372414529323578, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09171507507562637, "sampling/sampling_logp_difference/max": 3.1559271812438965, "sampling/importance_sampling_ratio/min": 0.04259888827800751, "sampling/importance_sampling_ratio/mean": 0.9968604445457458, "sampling/importance_sampling_ratio/max": 1.7393983602523804, "entropy": 0.4064619895070791, "clip_ratio/low_mean": 0.03893055208027363, "clip_ratio/low_min": 0.03893055208027363, "clip_ratio/high_mean": 0.05360710807144642, "clip_ratio/high_max": 0.05360710807144642, "clip_ratio/region_mean": 0.09253766015172005, "reward_total_mean": 0.7622486352920532, "reward_meter_mean": 0.9817661046981812, "reward_meter_std": 0.014558358117938042, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.17251639068126678, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8345391750335693, "reward_repeat_soft_std": 0.057019129395484924, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.1249857097864151, "reward_total_composite_mean": 0.7622486352920532, "reward_total_composite_std": 0.06372414529323578} {"timestamp_utc": "2026-04-13T03:12:14Z", "mode": "train", "global_step": 1944, "epoch": 0.1952787543947765, "loss": -0.0639, "grad_norm": 5.540454864501953, "learning_rate": 4.112121212121212e-06, "num_tokens": 3551840.0, "completions/mean_length": 145.0, "completions/min_length": 114.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 145.0, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.9827594757080078, "rewards/meter/std": 0.00882452167570591, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.569091796875, "rewards/repeat_soft/std": 0.12876377999782562, "rewards/judge_quality/mean": 0.3137499690055847, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7395259141921997, "rewards/total_composite/std": 0.040581319481134415, "reward": 0.7395259141921997, "reward_std": 0.04058130830526352, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06190090253949165, "sampling/sampling_logp_difference/max": 5.427993297576904, "sampling/importance_sampling_ratio/min": 0.004391900263726711, "sampling/importance_sampling_ratio/mean": 1.0020921230316162, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23555809631943703, "clip_ratio/low_mean": 0.0227449806407094, "clip_ratio/low_min": 0.0227449806407094, "clip_ratio/high_mean": 0.033464972861111164, "clip_ratio/high_max": 0.033464972861111164, "clip_ratio/region_mean": 0.056209953501820564, "reward_total_mean": 0.7395259141921997, "reward_meter_mean": 0.9827594757080078, "reward_meter_std": 0.00882452167570591, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.569091796875, "reward_repeat_soft_std": 0.12876377999782562, "reward_judge_quality_mean": 0.3137499690055847, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7395259141921997, "reward_total_composite_std": 0.040581319481134415} {"timestamp_utc": "2026-04-13T03:12:22Z", "mode": "train", "global_step": 1945, "epoch": 0.1953792064289302, "loss": 0.061, "grad_norm": 9.50792121887207, "learning_rate": 4.10909090909091e-06, "num_tokens": 3554040.0, "completions/mean_length": 112.0, "completions/min_length": 96.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.0, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.8186166882514954, "rewards/meter/std": 0.1986137479543686, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8772642612457275, "rewards/repeat_soft/std": 0.06695282459259033, "rewards/judge_quality/mean": 0.19624999165534973, "rewards/judge_quality/std": 0.09694438427686691, "rewards/total_composite/mean": 0.6587289571762085, "rewards/total_composite/std": 0.08277526497840881, "reward": 0.6587289571762085, "reward_std": 0.08277525007724762, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11078065633773804, "sampling/sampling_logp_difference/max": 3.335390090942383, "sampling/importance_sampling_ratio/min": 0.035600695759058, "sampling/importance_sampling_ratio/mean": 0.9932181239128113, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3933475650846958, "clip_ratio/low_mean": 0.03236268926411867, "clip_ratio/low_min": 0.03236268926411867, "clip_ratio/high_mean": 0.080354287289083, "clip_ratio/high_max": 0.080354287289083, "clip_ratio/region_mean": 0.11271697655320168, "reward_total_mean": 0.6587289571762085, "reward_meter_mean": 0.8186166882514954, "reward_meter_std": 0.1986137479543686, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8772642612457275, "reward_repeat_soft_std": 0.06695282459259033, "reward_judge_quality_mean": 0.19624999165534973, "reward_judge_quality_std": 0.09694438427686691, "reward_total_composite_mean": 0.6587289571762085, "reward_total_composite_std": 0.08277526497840881} {"timestamp_utc": "2026-04-13T03:12:30Z", "mode": "train", "global_step": 1946, "epoch": 0.19547965846308388, "loss": 0.0241, "grad_norm": 4.713191509246826, "learning_rate": 4.106060606060606e-06, "num_tokens": 3556374.0, "completions/mean_length": 109.75, "completions/min_length": 102.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.75, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.9543434977531433, "rewards/meter/std": 0.057896688580513, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8216732740402222, "rewards/repeat_soft/std": 0.09508197754621506, "rewards/judge_quality/mean": 0.3187499940395355, "rewards/judge_quality/std": 0.1397382616996765, "rewards/total_composite/mean": 0.7509969472885132, "rewards/total_composite/std": 0.054178424179553986, "reward": 0.7509969472885132, "reward_std": 0.05417841672897339, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07048658281564713, "sampling/sampling_logp_difference/max": 1.9658641815185547, "sampling/importance_sampling_ratio/min": 0.14003482460975647, "sampling/importance_sampling_ratio/mean": 1.0069684982299805, "sampling/importance_sampling_ratio/max": 1.9216768741607666, "entropy": 0.42542802169919014, "clip_ratio/low_mean": 0.028488762443885207, "clip_ratio/low_min": 0.028488762443885207, "clip_ratio/high_mean": 0.03253575274720788, "clip_ratio/high_max": 0.03253575274720788, "clip_ratio/region_mean": 0.06102451519109309, "reward_total_mean": 0.7509969472885132, "reward_meter_mean": 0.9543434977531433, "reward_meter_std": 0.057896688580513, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8216732740402222, "reward_repeat_soft_std": 0.09508197754621506, "reward_judge_quality_mean": 0.3187499940395355, "reward_judge_quality_std": 0.1397382616996765, "reward_total_composite_mean": 0.7509969472885132, "reward_total_composite_std": 0.054178424179553986} {"timestamp_utc": "2026-04-13T03:12:42Z", "mode": "train", "global_step": 1947, "epoch": 0.19558011049723756, "loss": -0.1319, "grad_norm": 3.2462267875671387, "learning_rate": 4.103030303030303e-06, "num_tokens": 3558332.0, "completions/mean_length": 134.75, "completions/min_length": 74.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 80.85714721679688, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.8448448777198792, "rewards/meter/std": 0.2612116038799286, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.928476870059967, "rewards/repeat_soft/std": 0.0668054148554802, "rewards/judge_quality/mean": 0.26875001192092896, "rewards/judge_quality/std": 0.1412634402513504, "rewards/total_composite/mean": 0.6942778825759888, "rewards/total_composite/std": 0.16379673779010773, "reward": 0.6942778825759888, "reward_std": 0.16379673779010773, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09358781576156616, "sampling/sampling_logp_difference/max": 2.2781291007995605, "sampling/importance_sampling_ratio/min": 0.10247574746608734, "sampling/importance_sampling_ratio/mean": 1.0115177631378174, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5405257530510426, "clip_ratio/low_mean": 0.01957831345498562, "clip_ratio/low_min": 0.01957831345498562, "clip_ratio/high_mean": 0.06687025958672166, "clip_ratio/high_max": 0.06687025958672166, "clip_ratio/region_mean": 0.08644857304170728, "reward_total_mean": 0.6942778825759888, "reward_meter_mean": 0.8448448777198792, "reward_meter_std": 0.2612116038799286, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.928476870059967, "reward_repeat_soft_std": 0.0668054148554802, "reward_judge_quality_mean": 0.26875001192092896, "reward_judge_quality_std": 0.1412634402513504, "reward_total_composite_mean": 0.6942778825759888, "reward_total_composite_std": 0.16379673779010773} {"timestamp_utc": "2026-04-13T03:12:54Z", "mode": "train", "global_step": 1948, "epoch": 0.19568056253139127, "loss": -0.1299, "grad_norm": 4.988866806030273, "learning_rate": 4.1e-06, "num_tokens": 3560116.0, "completions/mean_length": 124.0, "completions/min_length": 57.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 68.5714340209961, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.614212155342102, "rewards/meter/std": 0.4052046239376068, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9658875465393066, "rewards/repeat_soft/std": 0.037460070103406906, "rewards/judge_quality/mean": 0.19624999165534973, "rewards/judge_quality/std": 0.09694438427686691, "rewards/total_composite/mean": 0.5724841952323914, "rewards/total_composite/std": 0.18798024952411652, "reward": 0.5724841952323914, "reward_std": 0.18798024952411652, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10462155193090439, "sampling/sampling_logp_difference/max": 1.5934197902679443, "sampling/importance_sampling_ratio/min": 0.2032294124364853, "sampling/importance_sampling_ratio/mean": 1.0116323232650757, "sampling/importance_sampling_ratio/max": 1.8977596759796143, "entropy": 0.410679679363966, "clip_ratio/low_mean": 0.0389389768242836, "clip_ratio/low_min": 0.0389389768242836, "clip_ratio/high_mean": 0.042145861312747, "clip_ratio/high_max": 0.042145861312747, "clip_ratio/region_mean": 0.0810848381370306, "reward_total_mean": 0.5724841952323914, "reward_meter_mean": 0.614212155342102, "reward_meter_std": 0.4052046239376068, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9658875465393066, "reward_repeat_soft_std": 0.037460070103406906, "reward_judge_quality_mean": 0.19624999165534973, "reward_judge_quality_std": 0.09694438427686691, "reward_total_composite_mean": 0.5724841952323914, "reward_total_composite_std": 0.18798024952411652} {"timestamp_utc": "2026-04-13T03:13:00Z", "mode": "train", "global_step": 1949, "epoch": 0.19578101456554495, "loss": 0.0302, "grad_norm": 10.21060562133789, "learning_rate": 4.096969696969697e-06, "num_tokens": 3562020.0, "completions/mean_length": 63.0, "completions/min_length": 59.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9760424494743347, "rewards/meter/std": 0.030460350215435028, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9627712965011597, "rewards/repeat_soft/std": 0.02374342828989029, "rewards/judge_quality/mean": 0.38874998688697815, "rewards/judge_quality/std": 0.08675704896450043, "rewards/total_composite/mean": 0.8021212816238403, "rewards/total_composite/std": 0.03622322902083397, "reward": 0.8021212816238403, "reward_std": 0.036223217844963074, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09309766441583633, "sampling/sampling_logp_difference/max": 1.1224713325500488, "sampling/importance_sampling_ratio/min": 0.32547444105148315, "sampling/importance_sampling_ratio/mean": 1.004652976989746, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49071964249014854, "clip_ratio/low_mean": 0.01171307498589158, "clip_ratio/low_min": 0.01171307498589158, "clip_ratio/high_mean": 0.0807400681078434, "clip_ratio/high_max": 0.0807400681078434, "clip_ratio/region_mean": 0.09245314309373498, "reward_total_mean": 0.8021212816238403, "reward_meter_mean": 0.9760424494743347, "reward_meter_std": 0.030460350215435028, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9627712965011597, "reward_repeat_soft_std": 0.02374342828989029, "reward_judge_quality_mean": 0.38874998688697815, "reward_judge_quality_std": 0.08675704896450043, "reward_total_composite_mean": 0.8021212816238403, "reward_total_composite_std": 0.03622322902083397} {"timestamp_utc": "2026-04-13T03:13:08Z", "mode": "train", "global_step": 1950, "epoch": 0.19588146659969866, "loss": 0.0178, "grad_norm": 6.411823749542236, "learning_rate": 4.093939393939394e-06, "num_tokens": 3564347.0, "completions/mean_length": 120.875, "completions/min_length": 113.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.875, "completions/min_terminated_length": 113.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.6641327142715454, "rewards/meter/std": 0.38974547386169434, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9320442080497742, "rewards/repeat_soft/std": 0.022080201655626297, "rewards/judge_quality/mean": 0.16250000894069672, "rewards/judge_quality/std": 0.0353553369641304, "rewards/total_composite/mean": 0.5845641493797302, "rewards/total_composite/std": 0.17603452503681183, "reward": 0.5845641493797302, "reward_std": 0.17603452503681183, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09629218280315399, "sampling/sampling_logp_difference/max": 1.887526035308838, "sampling/importance_sampling_ratio/min": 0.15144601464271545, "sampling/importance_sampling_ratio/mean": 1.0067353248596191, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5186042338609695, "clip_ratio/low_mean": 0.043355807196348906, "clip_ratio/low_min": 0.043355807196348906, "clip_ratio/high_mean": 0.05341572780162096, "clip_ratio/high_max": 0.05341572780162096, "clip_ratio/region_mean": 0.09677153499796987, "reward_total_mean": 0.5845641493797302, "reward_meter_mean": 0.6641327142715454, "reward_meter_std": 0.38974547386169434, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9320442080497742, "reward_repeat_soft_std": 0.022080201655626297, "reward_judge_quality_mean": 0.16250000894069672, "reward_judge_quality_std": 0.0353553369641304, "reward_total_composite_mean": 0.5845641493797302, "reward_total_composite_std": 0.17603452503681183} {"timestamp_utc": "2026-04-13T03:14:05Z", "mode": "eval", "global_step": 1950, "epoch": 0.19588146659969866, "eval_loss": NaN, "eval_runtime": 56.9527, "eval_samples_per_second": 1.405, "eval_steps_per_second": 0.176, "eval_num_tokens": 3564347.0, "eval_completions/mean_length": 115.3375, "eval_completions/min_length": 52.4, "eval_completions/max_length": 211.6, "eval_completions/clipped_ratio": 0.0125, "eval_completions/mean_terminated_length": 110.33035736083984, "eval_completions/min_terminated_length": 52.4, "eval_completions/max_terminated_length": 177.4, "eval_rewards/meter/mean": 0.8197983562946319, "eval_rewards/meter/std": 0.2503858106210828, "eval_rewards/count_adherence/mean": 0.9691666722297668, "eval_rewards/count_adherence/std": 0.06022390499711037, "eval_rewards/hard_gate/mean": 0.9875, "eval_rewards/hard_gate/std": 0.03535533845424652, "eval_rewards/repeat_soft/mean": 0.8630496859550476, "eval_rewards/repeat_soft/std": 0.09276245608925819, "eval_rewards/judge_quality/mean": 0.28150000125169755, "eval_rewards/judge_quality/std": 0.13229885324835777, "eval_rewards/total_composite/mean": 0.6771972417831421, "eval_rewards/total_composite/std": 0.139620803296566, "eval_reward": 0.6771972417831421, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.04132597073912621, "eval_sampling/sampling_logp_difference/max": 0.9355854034423828, "eval_sampling/importance_sampling_ratio/min": 0.39677918553352354, "eval_sampling/importance_sampling_ratio/mean": 1.0077006340026855, "eval_sampling/importance_sampling_ratio/max": 1.3089853763580321, "eval_entropy": 0.44099169969558716, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6771972417831421, "eval_reward_meter_mean": 0.8197983562946319, "eval_reward_meter_std": 0.2503858106210828, "eval_reward_count_adherence_mean": 0.9691666722297668, "eval_reward_count_adherence_std": 0.06022390499711037, "eval_reward_hard_gate_mean": 0.9875, "eval_reward_hard_gate_std": 0.03535533845424652, "eval_reward_repeat_soft_mean": 0.8630496859550476, "eval_reward_repeat_soft_std": 0.09276245608925819, "eval_reward_judge_quality_mean": 0.28150000125169755, "eval_reward_judge_quality_std": 0.13229885324835777, "eval_reward_total_composite_mean": 0.6771972417831421, "eval_reward_total_composite_std": 0.139620803296566} {"timestamp_utc": "2026-04-13T03:14:15Z", "mode": "train", "global_step": 1951, "epoch": 0.19598191863385234, "loss": 0.0872, "grad_norm": 9.530619621276855, "learning_rate": 4.0909090909090915e-06, "num_tokens": 3566140.0, "completions/mean_length": 71.125, "completions/min_length": 64.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.125, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9851779937744141, "rewards/meter/std": 0.006869742181152105, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9398838877677917, "rewards/repeat_soft/std": 0.05847184360027313, "rewards/judge_quality/mean": 0.2824999690055847, "rewards/judge_quality/std": 0.12578324973583221, "rewards/total_composite/mean": 0.7720685005187988, "rewards/total_composite/std": 0.044727619737386703, "reward": 0.7720685005187988, "reward_std": 0.044727619737386703, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10405370593070984, "sampling/sampling_logp_difference/max": 1.963963270187378, "sampling/importance_sampling_ratio/min": 0.14030127227306366, "sampling/importance_sampling_ratio/mean": 1.0109457969665527, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6320488378405571, "clip_ratio/low_mean": 0.05380029324442148, "clip_ratio/low_min": 0.05380029324442148, "clip_ratio/high_mean": 0.033799865283071995, "clip_ratio/high_max": 0.033799865283071995, "clip_ratio/region_mean": 0.08760015852749348, "reward_total_mean": 0.7720685005187988, "reward_meter_mean": 0.9851779937744141, "reward_meter_std": 0.006869742181152105, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9398838877677917, "reward_repeat_soft_std": 0.05847184360027313, "reward_judge_quality_mean": 0.2824999690055847, "reward_judge_quality_std": 0.12578324973583221, "reward_total_composite_mean": 0.7720685005187988, "reward_total_composite_std": 0.044727619737386703} {"timestamp_utc": "2026-04-13T03:14:21Z", "mode": "train", "global_step": 1952, "epoch": 0.19608237066800602, "loss": 0.0366, "grad_norm": 10.178441047668457, "learning_rate": 4.087878787878789e-06, "num_tokens": 3567771.0, "completions/mean_length": 49.875, "completions/min_length": 47.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.875, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.9203057885169983, "rewards/meter/std": 0.13779860734939575, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9754425287246704, "rewards/repeat_soft/std": 0.012847279198467731, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7888068556785583, "rewards/total_composite/std": 0.060281943529844284, "reward": 0.7888068556785583, "reward_std": 0.06028194725513458, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0904785543680191, "sampling/sampling_logp_difference/max": 2.7055983543395996, "sampling/importance_sampling_ratio/min": 0.06683032214641571, "sampling/importance_sampling_ratio/mean": 0.9884557723999023, "sampling/importance_sampling_ratio/max": 1.8735696077346802, "entropy": 0.37962542846798897, "clip_ratio/low_mean": 0.01256257202476263, "clip_ratio/low_min": 0.01256257202476263, "clip_ratio/high_mean": 0.06768860714510083, "clip_ratio/high_max": 0.06768860714510083, "clip_ratio/region_mean": 0.08025117916986346, "reward_total_mean": 0.7888068556785583, "reward_meter_mean": 0.9203057885169983, "reward_meter_std": 0.13779860734939575, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9754425287246704, "reward_repeat_soft_std": 0.012847279198467731, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7888068556785583, "reward_total_composite_std": 0.060281943529844284} {"timestamp_utc": "2026-04-13T03:14:28Z", "mode": "train", "global_step": 1953, "epoch": 0.19618282270215973, "loss": 0.0141, "grad_norm": 9.179468154907227, "learning_rate": 4.084848484848485e-06, "num_tokens": 3569551.0, "completions/mean_length": 66.5, "completions/min_length": 59.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.965185284614563, "rewards/meter/std": 0.044416215270757675, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9016900062561035, "rewards/repeat_soft/std": 0.054999418556690216, "rewards/judge_quality/mean": 0.3387500047683716, "rewards/judge_quality/std": 0.13292719423770905, "rewards/total_composite/mean": 0.7761273980140686, "rewards/total_composite/std": 0.03659522905945778, "reward": 0.7761273980140686, "reward_std": 0.03659522905945778, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08087209612131119, "sampling/sampling_logp_difference/max": 1.2893562316894531, "sampling/importance_sampling_ratio/min": 0.2754480242729187, "sampling/importance_sampling_ratio/mean": 0.9989621043205261, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3753427043557167, "clip_ratio/low_mean": 0.03309682570397854, "clip_ratio/low_min": 0.03309682570397854, "clip_ratio/high_mean": 0.043477474711835384, "clip_ratio/high_max": 0.043477474711835384, "clip_ratio/region_mean": 0.07657430041581392, "reward_total_mean": 0.7761273980140686, "reward_meter_mean": 0.965185284614563, "reward_meter_std": 0.044416215270757675, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9016900062561035, "reward_repeat_soft_std": 0.054999418556690216, "reward_judge_quality_mean": 0.3387500047683716, "reward_judge_quality_std": 0.13292719423770905, "reward_total_composite_mean": 0.7761273980140686, "reward_total_composite_std": 0.03659522905945778} {"timestamp_utc": "2026-04-13T03:14:35Z", "mode": "train", "global_step": 1954, "epoch": 0.1962832747363134, "loss": 0.0017, "grad_norm": 13.132867813110352, "learning_rate": 4.081818181818182e-06, "num_tokens": 3571515.0, "completions/mean_length": 74.5, "completions/min_length": 66.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.5, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.8693011403083801, "rewards/meter/std": 0.22581082582473755, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9346792101860046, "rewards/repeat_soft/std": 0.04577529430389404, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.12631450593471527, "rewards/total_composite/mean": 0.7340284585952759, "rewards/total_composite/std": 0.09989648312330246, "reward": 0.7340284585952759, "reward_std": 0.09989648312330246, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10461217910051346, "sampling/sampling_logp_difference/max": 1.7054309844970703, "sampling/importance_sampling_ratio/min": 0.18169406056404114, "sampling/importance_sampling_ratio/mean": 1.0137580633163452, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5204013176262379, "clip_ratio/low_mean": 0.026504757814109325, "clip_ratio/low_min": 0.026504757814109325, "clip_ratio/high_mean": 0.06235684780403972, "clip_ratio/high_max": 0.06235684780403972, "clip_ratio/region_mean": 0.08886160561814904, "reward_total_mean": 0.7340284585952759, "reward_meter_mean": 0.8693011403083801, "reward_meter_std": 0.22581082582473755, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9346792101860046, "reward_repeat_soft_std": 0.04577529430389404, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.12631450593471527, "reward_total_composite_mean": 0.7340284585952759, "reward_total_composite_std": 0.09989648312330246} {"timestamp_utc": "2026-04-13T03:14:46Z", "mode": "train", "global_step": 1955, "epoch": 0.19638372677046712, "loss": -0.141, "grad_norm": 2.146821975708008, "learning_rate": 4.07878787878788e-06, "num_tokens": 3573281.0, "completions/mean_length": 188.75, "completions/min_length": 68.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 81.0, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.8637641668319702, "rewards/meter/std": 0.29630187153816223, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9289455413818359, "rewards/repeat_soft/std": 0.07835699617862701, "rewards/judge_quality/mean": 0.22624999284744263, "rewards/judge_quality/std": 0.16569657623767853, "rewards/total_composite/mean": 0.5231028199195862, "rewards/total_composite/std": 0.35661983489990234, "reward": 0.5231028199195862, "reward_std": 0.35661983489990234, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09929997473955154, "sampling/sampling_logp_difference/max": 1.3953475952148438, "sampling/importance_sampling_ratio/min": 0.24774691462516785, "sampling/importance_sampling_ratio/mean": 1.0181068181991577, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.42986229062080383, "clip_ratio/low_mean": 0.012362637557089329, "clip_ratio/low_min": 0.012362637557089329, "clip_ratio/high_mean": 0.055148005951195955, "clip_ratio/high_max": 0.055148005951195955, "clip_ratio/region_mean": 0.06751064350828528, "reward_total_mean": 0.5231028199195862, "reward_meter_mean": 0.8637641668319702, "reward_meter_std": 0.29630187153816223, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9289455413818359, "reward_repeat_soft_std": 0.07835699617862701, "reward_judge_quality_mean": 0.22624999284744263, "reward_judge_quality_std": 0.16569657623767853, "reward_total_composite_mean": 0.5231028199195862, "reward_total_composite_std": 0.35661983489990234} {"timestamp_utc": "2026-04-13T03:14:53Z", "mode": "train", "global_step": 1956, "epoch": 0.1964841788046208, "loss": 0.0301, "grad_norm": 11.77906322479248, "learning_rate": 4.075757575757576e-06, "num_tokens": 3575025.0, "completions/mean_length": 72.0, "completions/min_length": 67.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9242136478424072, "rewards/meter/std": 0.14695116877555847, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9016977548599243, "rewards/repeat_soft/std": 0.04889515042304993, "rewards/judge_quality/mean": 0.23749999701976776, "rewards/judge_quality/std": 0.13562026619911194, "rewards/total_composite/mean": 0.7273159027099609, "rewards/total_composite/std": 0.05501732602715492, "reward": 0.7273159027099609, "reward_std": 0.055017322301864624, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10805526375770569, "sampling/sampling_logp_difference/max": 2.0816311836242676, "sampling/importance_sampling_ratio/min": 0.12472659349441528, "sampling/importance_sampling_ratio/mean": 1.001023769378662, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5515466779470444, "clip_ratio/low_mean": 0.049058311618864536, "clip_ratio/low_min": 0.049058311618864536, "clip_ratio/high_mean": 0.044816031120717525, "clip_ratio/high_max": 0.044816031120717525, "clip_ratio/region_mean": 0.09387434273958206, "reward_total_mean": 0.7273159027099609, "reward_meter_mean": 0.9242136478424072, "reward_meter_std": 0.14695116877555847, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9016977548599243, "reward_repeat_soft_std": 0.04889515042304993, "reward_judge_quality_mean": 0.23749999701976776, "reward_judge_quality_std": 0.13562026619911194, "reward_total_composite_mean": 0.7273159027099609, "reward_total_composite_std": 0.05501732602715492} {"timestamp_utc": "2026-04-13T03:15:01Z", "mode": "train", "global_step": 1957, "epoch": 0.19658463083877448, "loss": 0.0504, "grad_norm": 5.546024799346924, "learning_rate": 4.072727272727273e-06, "num_tokens": 3578075.0, "completions/mean_length": 185.25, "completions/min_length": 165.0, "completions/max_length": 202.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 185.25, "completions/min_terminated_length": 165.0, "completions/max_terminated_length": 202.0, "rewards/meter/mean": 0.8261789083480835, "rewards/meter/std": 0.24118322134017944, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7343109250068665, "rewards/repeat_soft/std": 0.14047551155090332, "rewards/judge_quality/mean": 0.16250000894069672, "rewards/judge_quality/std": 0.0353553369641304, "rewards/total_composite/mean": 0.6189616322517395, "rewards/total_composite/std": 0.12242996692657471, "reward": 0.6189616322517395, "reward_std": 0.1224299743771553, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07458855211734772, "sampling/sampling_logp_difference/max": 3.2073822021484375, "sampling/importance_sampling_ratio/min": 0.04046240076422691, "sampling/importance_sampling_ratio/mean": 1.006753921508789, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3120560385286808, "clip_ratio/low_mean": 0.017757622757926583, "clip_ratio/low_min": 0.017757622757926583, "clip_ratio/high_mean": 0.04836894292384386, "clip_ratio/high_max": 0.04836894292384386, "clip_ratio/region_mean": 0.06612656568177044, "reward_total_mean": 0.6189616322517395, "reward_meter_mean": 0.8261789083480835, "reward_meter_std": 0.24118322134017944, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7343109250068665, "reward_repeat_soft_std": 0.14047551155090332, "reward_judge_quality_mean": 0.16250000894069672, "reward_judge_quality_std": 0.0353553369641304, "reward_total_composite_mean": 0.6189616322517395, "reward_total_composite_std": 0.12242996692657471} {"timestamp_utc": "2026-04-13T03:15:08Z", "mode": "train", "global_step": 1958, "epoch": 0.19668508287292819, "loss": -0.0194, "grad_norm": 13.758614540100098, "learning_rate": 4.0696969696969706e-06, "num_tokens": 3579647.0, "completions/mean_length": 42.5, "completions/min_length": 29.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.5, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.8493857979774475, "rewards/meter/std": 0.11957097053527832, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9810284972190857, "rewards/repeat_soft/std": 0.017662055790424347, "rewards/judge_quality/mean": 0.39249998331069946, "rewards/judge_quality/std": 0.08892211318016052, "rewards/total_composite/mean": 0.7480764985084534, "rewards/total_composite/std": 0.05417594686150551, "reward": 0.7480764985084534, "reward_std": 0.054175958037376404, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.117611363530159, "sampling/sampling_logp_difference/max": 1.966352939605713, "sampling/importance_sampling_ratio/min": 0.13996639847755432, "sampling/importance_sampling_ratio/mean": 1.0221763849258423, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6056117415428162, "clip_ratio/low_mean": 0.06296979822218418, "clip_ratio/low_min": 0.06296979822218418, "clip_ratio/high_mean": 0.04154693498276174, "clip_ratio/high_max": 0.04154693498276174, "clip_ratio/region_mean": 0.10451673320494592, "reward_total_mean": 0.7480764985084534, "reward_meter_mean": 0.8493857979774475, "reward_meter_std": 0.11957097053527832, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9810284972190857, "reward_repeat_soft_std": 0.017662055790424347, "reward_judge_quality_mean": 0.39249998331069946, "reward_judge_quality_std": 0.08892211318016052, "reward_total_composite_mean": 0.7480764985084534, "reward_total_composite_std": 0.05417594686150551} {"timestamp_utc": "2026-04-13T03:15:14Z", "mode": "train", "global_step": 1959, "epoch": 0.19678553490708187, "loss": 0.0509, "grad_norm": 11.691507339477539, "learning_rate": 4.066666666666667e-06, "num_tokens": 3581287.0, "completions/mean_length": 48.0, "completions/min_length": 44.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.0, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.8371472358703613, "rewards/meter/std": 0.15107029676437378, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9524387717247009, "rewards/repeat_soft/std": 0.03612814098596573, "rewards/judge_quality/mean": 0.5149999856948853, "rewards/judge_quality/std": 0.2676885426044464, "rewards/total_composite/mean": 0.7764601111412048, "rewards/total_composite/std": 0.0851791575551033, "reward": 0.7764601111412048, "reward_std": 0.0851791724562645, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10554791986942291, "sampling/sampling_logp_difference/max": 1.8329894542694092, "sampling/importance_sampling_ratio/min": 0.15993474423885345, "sampling/importance_sampling_ratio/mean": 1.0106855630874634, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5208171680569649, "clip_ratio/low_mean": 0.047390651889145374, "clip_ratio/low_min": 0.047390651889145374, "clip_ratio/high_mean": 0.07364790048450232, "clip_ratio/high_max": 0.07364790048450232, "clip_ratio/region_mean": 0.12103855237364769, "reward_total_mean": 0.7764601111412048, "reward_meter_mean": 0.8371472358703613, "reward_meter_std": 0.15107029676437378, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9524387717247009, "reward_repeat_soft_std": 0.03612814098596573, "reward_judge_quality_mean": 0.5149999856948853, "reward_judge_quality_std": 0.2676885426044464, "reward_total_composite_mean": 0.7764601111412048, "reward_total_composite_std": 0.0851791575551033} {"timestamp_utc": "2026-04-13T03:15:20Z", "mode": "train", "global_step": 1960, "epoch": 0.19688598694123555, "loss": 0.0617, "grad_norm": 12.733287811279297, "learning_rate": 4.063636363636364e-06, "num_tokens": 3582777.0, "completions/mean_length": 38.25, "completions/min_length": 35.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.25, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.743402361869812, "rewards/meter/std": 0.2644272744655609, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9341490864753723, "rewards/repeat_soft/std": 0.04466065764427185, "rewards/judge_quality/mean": 0.3512499928474426, "rewards/judge_quality/std": 0.11630470305681229, "rewards/total_composite/mean": 0.6833209991455078, "rewards/total_composite/std": 0.12097742408514023, "reward": 0.6833209991455078, "reward_std": 0.12097743153572083, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10879290848970413, "sampling/sampling_logp_difference/max": 0.8389580249786377, "sampling/importance_sampling_ratio/min": 0.4321605861186981, "sampling/importance_sampling_ratio/mean": 1.0270631313323975, "sampling/importance_sampling_ratio/max": 1.902855396270752, "entropy": 0.7535847648978233, "clip_ratio/low_mean": 0.06168355792760849, "clip_ratio/low_min": 0.06168355792760849, "clip_ratio/high_mean": 0.056642450392246246, "clip_ratio/high_max": 0.056642450392246246, "clip_ratio/region_mean": 0.11832600831985474, "reward_total_mean": 0.6833209991455078, "reward_meter_mean": 0.743402361869812, "reward_meter_std": 0.2644272744655609, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9341490864753723, "reward_repeat_soft_std": 0.04466065764427185, "reward_judge_quality_mean": 0.3512499928474426, "reward_judge_quality_std": 0.11630470305681229, "reward_total_composite_mean": 0.6833209991455078, "reward_total_composite_std": 0.12097742408514023} {"timestamp_utc": "2026-04-13T03:15:26Z", "mode": "train", "global_step": 1961, "epoch": 0.19698643897538926, "loss": 0.0912, "grad_norm": 60.9083366394043, "learning_rate": 4.060606060606061e-06, "num_tokens": 3584171.0, "completions/mean_length": 23.25, "completions/min_length": 19.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.25, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.5597894191741943, "rewards/meter/std": 0.3953060507774353, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.6200302839279175, "rewards/total_composite/std": 0.19034765660762787, "reward": 0.6200302839279175, "reward_std": 0.19034767150878906, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11257188022136688, "sampling/sampling_logp_difference/max": 1.265604019165039, "sampling/importance_sampling_ratio/min": 0.2820688784122467, "sampling/importance_sampling_ratio/mean": 1.0138682126998901, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5300300642848015, "clip_ratio/low_mean": 0.04195804335176945, "clip_ratio/low_min": 0.04195804335176945, "clip_ratio/high_mean": 0.06848236033692956, "clip_ratio/high_max": 0.06848236033692956, "clip_ratio/region_mean": 0.11044040368869901, "reward_total_mean": 0.6200302839279175, "reward_meter_mean": 0.5597894191741943, "reward_meter_std": 0.3953060507774353, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.6200302839279175, "reward_total_composite_std": 0.19034765660762787} {"timestamp_utc": "2026-04-13T03:15:33Z", "mode": "train", "global_step": 1962, "epoch": 0.19708689100954294, "loss": 0.0246, "grad_norm": 11.05737018585205, "learning_rate": 4.057575757575758e-06, "num_tokens": 3586007.0, "completions/mean_length": 66.5, "completions/min_length": 64.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9529627561569214, "rewards/meter/std": 0.029169093817472458, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9389608502388, "rewards/repeat_soft/std": 0.05336339399218559, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7987293004989624, "rewards/total_composite/std": 0.010347177274525166, "reward": 0.7987293004989624, "reward_std": 0.010347179137170315, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09377987682819366, "sampling/sampling_logp_difference/max": 1.4631645679473877, "sampling/importance_sampling_ratio/min": 0.23150251805782318, "sampling/importance_sampling_ratio/mean": 1.0096604824066162, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5214551240205765, "clip_ratio/low_mean": 0.012500000186264515, "clip_ratio/low_min": 0.012500000186264515, "clip_ratio/high_mean": 0.07548737479373813, "clip_ratio/high_max": 0.07548737479373813, "clip_ratio/region_mean": 0.08798737498000264, "reward_total_mean": 0.7987293004989624, "reward_meter_mean": 0.9529627561569214, "reward_meter_std": 0.029169093817472458, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9389608502388, "reward_repeat_soft_std": 0.05336339399218559, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7987293004989624, "reward_total_composite_std": 0.010347177274525166} {"timestamp_utc": "2026-04-13T03:15:39Z", "mode": "train", "global_step": 1963, "epoch": 0.19718734304369664, "loss": 0.0065, "grad_norm": 18.38215446472168, "learning_rate": 4.054545454545455e-06, "num_tokens": 3587630.0, "completions/mean_length": 42.875, "completions/min_length": 38.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.875, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.8521711826324463, "rewards/meter/std": 0.28859567642211914, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9609649777412415, "rewards/repeat_soft/std": 0.02794629894196987, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7611985206604004, "rewards/total_composite/std": 0.13361522555351257, "reward": 0.7611985206604004, "reward_std": 0.13361521065235138, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09487352520227432, "sampling/sampling_logp_difference/max": 1.5480825901031494, "sampling/importance_sampling_ratio/min": 0.21265532076358795, "sampling/importance_sampling_ratio/mean": 0.9946743249893188, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39602838084101677, "clip_ratio/low_mean": 0.02187499962747097, "clip_ratio/low_min": 0.02187499962747097, "clip_ratio/high_mean": 0.097631074488163, "clip_ratio/high_max": 0.097631074488163, "clip_ratio/region_mean": 0.11950607411563396, "reward_total_mean": 0.7611985206604004, "reward_meter_mean": 0.8521711826324463, "reward_meter_std": 0.28859567642211914, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9609649777412415, "reward_repeat_soft_std": 0.02794629894196987, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7611985206604004, "reward_total_composite_std": 0.13361522555351257} {"timestamp_utc": "2026-04-13T03:15:46Z", "mode": "train", "global_step": 1964, "epoch": 0.19728779507785033, "loss": 0.0391, "grad_norm": 6.343920707702637, "learning_rate": 4.0515151515151516e-06, "num_tokens": 3589844.0, "completions/mean_length": 112.75, "completions/min_length": 107.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.75, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.8444241285324097, "rewards/meter/std": 0.24861839413642883, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9324159622192383, "rewards/repeat_soft/std": 0.0385960228741169, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7428574562072754, "rewards/total_composite/std": 0.10806667804718018, "reward": 0.7428574562072754, "reward_std": 0.10806667059659958, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09189782291650772, "sampling/sampling_logp_difference/max": 1.5090837478637695, "sampling/importance_sampling_ratio/min": 0.22111248970031738, "sampling/importance_sampling_ratio/mean": 1.005723237991333, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45612039417028427, "clip_ratio/low_mean": 0.02028985507786274, "clip_ratio/low_min": 0.02028985507786274, "clip_ratio/high_mean": 0.056897740345448256, "clip_ratio/high_max": 0.056897740345448256, "clip_ratio/region_mean": 0.077187595423311, "reward_total_mean": 0.7428574562072754, "reward_meter_mean": 0.8444241285324097, "reward_meter_std": 0.24861839413642883, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9324159622192383, "reward_repeat_soft_std": 0.0385960228741169, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7428574562072754, "reward_total_composite_std": 0.10806667804718018} {"timestamp_utc": "2026-04-13T03:15:53Z", "mode": "train", "global_step": 1965, "epoch": 0.197388247112004, "loss": 0.0366, "grad_norm": 5.479944705963135, "learning_rate": 4.048484848484849e-06, "num_tokens": 3591902.0, "completions/mean_length": 76.25, "completions/min_length": 71.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.25, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9807966351509094, "rewards/meter/std": 0.005391569808125496, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8906934261322021, "rewards/repeat_soft/std": 0.04095250740647316, "rewards/judge_quality/mean": 0.1837500035762787, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.7355527877807617, "rewards/total_composite/std": 0.029530132189393044, "reward": 0.7355527877807617, "reward_std": 0.029530145227909088, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0851706862449646, "sampling/sampling_logp_difference/max": 1.7777376174926758, "sampling/importance_sampling_ratio/min": 0.16902010142803192, "sampling/importance_sampling_ratio/mean": 1.0068923234939575, "sampling/importance_sampling_ratio/max": 1.900407075881958, "entropy": 0.49146900326013565, "clip_ratio/low_mean": 0.051971081411466, "clip_ratio/low_min": 0.051971081411466, "clip_ratio/high_mean": 0.008802817203104496, "clip_ratio/high_max": 0.008802817203104496, "clip_ratio/region_mean": 0.0607738986145705, "reward_total_mean": 0.7355527877807617, "reward_meter_mean": 0.9807966351509094, "reward_meter_std": 0.005391569808125496, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8906934261322021, "reward_repeat_soft_std": 0.04095250740647316, "reward_judge_quality_mean": 0.1837500035762787, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.7355527877807617, "reward_total_composite_std": 0.029530132189393044} {"timestamp_utc": "2026-04-13T03:15:59Z", "mode": "train", "global_step": 1966, "epoch": 0.19748869914615771, "loss": 0.0095, "grad_norm": 14.102391242980957, "learning_rate": 4.045454545454546e-06, "num_tokens": 3593371.0, "completions/mean_length": 36.625, "completions/min_length": 33.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.625, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.5255882143974304, "rewards/meter/std": 0.39444053173065186, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9336788058280945, "rewards/repeat_soft/std": 0.05528785288333893, "rewards/judge_quality/mean": 0.4087499976158142, "rewards/judge_quality/std": 0.24706491827964783, "rewards/total_composite/mean": 0.6025075912475586, "rewards/total_composite/std": 0.19815507531166077, "reward": 0.6025075912475586, "reward_std": 0.19815504550933838, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13457396626472473, "sampling/sampling_logp_difference/max": 1.4217960834503174, "sampling/importance_sampling_ratio/min": 0.24128027260303497, "sampling/importance_sampling_ratio/mean": 1.0115740299224854, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6604181602597237, "clip_ratio/low_mean": 0.054496207274496555, "clip_ratio/low_min": 0.054496207274496555, "clip_ratio/high_mean": 0.020650585182011127, "clip_ratio/high_max": 0.020650585182011127, "clip_ratio/region_mean": 0.07514679245650768, "reward_total_mean": 0.6025075912475586, "reward_meter_mean": 0.5255882143974304, "reward_meter_std": 0.39444053173065186, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9336788058280945, "reward_repeat_soft_std": 0.05528785288333893, "reward_judge_quality_mean": 0.4087499976158142, "reward_judge_quality_std": 0.24706491827964783, "reward_total_composite_mean": 0.6025075912475586, "reward_total_composite_std": 0.19815507531166077} {"timestamp_utc": "2026-04-13T03:16:06Z", "mode": "train", "global_step": 1967, "epoch": 0.1975891511803114, "loss": 0.0285, "grad_norm": 13.731514930725098, "learning_rate": 4.0424242424242425e-06, "num_tokens": 3595140.0, "completions/mean_length": 65.125, "completions/min_length": 53.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.125, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.8633257150650024, "rewards/meter/std": 0.15666554868221283, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9220964312553406, "rewards/repeat_soft/std": 0.05571933090686798, "rewards/judge_quality/mean": 0.29249998927116394, "rewards/judge_quality/std": 0.12150836735963821, "rewards/total_composite/mean": 0.7184562683105469, "rewards/total_composite/std": 0.08906339854001999, "reward": 0.7184562683105469, "reward_std": 0.08906339854001999, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12613892555236816, "sampling/sampling_logp_difference/max": 2.101804733276367, "sampling/importance_sampling_ratio/min": 0.12223562598228455, "sampling/importance_sampling_ratio/mean": 0.9817524552345276, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5031076222658157, "clip_ratio/low_mean": 0.035660392604768276, "clip_ratio/low_min": 0.035660392604768276, "clip_ratio/high_mean": 0.08671625051647425, "clip_ratio/high_max": 0.08671625051647425, "clip_ratio/region_mean": 0.12237664312124252, "reward_total_mean": 0.7184562683105469, "reward_meter_mean": 0.8633257150650024, "reward_meter_std": 0.15666554868221283, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9220964312553406, "reward_repeat_soft_std": 0.05571933090686798, "reward_judge_quality_mean": 0.29249998927116394, "reward_judge_quality_std": 0.12150836735963821, "reward_total_composite_mean": 0.7184562683105469, "reward_total_composite_std": 0.08906339854001999} {"timestamp_utc": "2026-04-13T03:16:13Z", "mode": "train", "global_step": 1968, "epoch": 0.1976896032144651, "loss": 0.0825, "grad_norm": 12.044737815856934, "learning_rate": 4.03939393939394e-06, "num_tokens": 3596818.0, "completions/mean_length": 59.75, "completions/min_length": 56.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.75, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9749720096588135, "rewards/meter/std": 0.021912995725870132, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9474745988845825, "rewards/repeat_soft/std": 0.0534500814974308, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.803109884262085, "rewards/total_composite/std": 0.016200901940464973, "reward": 0.803109884262085, "reward_std": 0.016200894489884377, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09442979842424393, "sampling/sampling_logp_difference/max": 1.0041543245315552, "sampling/importance_sampling_ratio/min": 0.3663543164730072, "sampling/importance_sampling_ratio/mean": 1.0048452615737915, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4706795662641525, "clip_ratio/low_mean": 0.024738655425608158, "clip_ratio/low_min": 0.024738655425608158, "clip_ratio/high_mean": 0.06910828035324812, "clip_ratio/high_max": 0.06910828035324812, "clip_ratio/region_mean": 0.09384693577885628, "reward_total_mean": 0.803109884262085, "reward_meter_mean": 0.9749720096588135, "reward_meter_std": 0.021912995725870132, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9474745988845825, "reward_repeat_soft_std": 0.0534500814974308, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.803109884262085, "reward_total_composite_std": 0.016200901940464973} {"timestamp_utc": "2026-04-13T03:16:20Z", "mode": "train", "global_step": 1969, "epoch": 0.19779005524861878, "loss": 0.0551, "grad_norm": 6.704044342041016, "learning_rate": 4.036363636363637e-06, "num_tokens": 3599003.0, "completions/mean_length": 107.125, "completions/min_length": 94.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.125, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.9830548763275146, "rewards/meter/std": 0.007242096588015556, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9228017330169678, "rewards/repeat_soft/std": 0.04816256836056709, "rewards/judge_quality/mean": 0.2212499976158142, "rewards/judge_quality/std": 0.09433034062385559, "rewards/total_composite/mean": 0.7510298490524292, "rewards/total_composite/std": 0.029415355995297432, "reward": 0.7510298490524292, "reward_std": 0.02941535972058773, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10686644911766052, "sampling/sampling_logp_difference/max": 1.0721983909606934, "sampling/importance_sampling_ratio/min": 0.34225529432296753, "sampling/importance_sampling_ratio/mean": 1.0005683898925781, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6885115541517735, "clip_ratio/low_mean": 0.03153043892234564, "clip_ratio/low_min": 0.03153043892234564, "clip_ratio/high_mean": 0.05811917223036289, "clip_ratio/high_max": 0.05811917223036289, "clip_ratio/region_mean": 0.08964961115270853, "reward_total_mean": 0.7510298490524292, "reward_meter_mean": 0.9830548763275146, "reward_meter_std": 0.007242096588015556, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9228017330169678, "reward_repeat_soft_std": 0.04816256836056709, "reward_judge_quality_mean": 0.2212499976158142, "reward_judge_quality_std": 0.09433034062385559, "reward_total_composite_mean": 0.7510298490524292, "reward_total_composite_std": 0.029415355995297432} {"timestamp_utc": "2026-04-13T03:16:27Z", "mode": "train", "global_step": 1970, "epoch": 0.19789050728277247, "loss": -0.0005, "grad_norm": 8.065711975097656, "learning_rate": 4.033333333333333e-06, "num_tokens": 3600847.0, "completions/mean_length": 77.5, "completions/min_length": 68.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.5, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.6033499240875244, "rewards/meter/std": 0.3766990900039673, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9360681176185608, "rewards/repeat_soft/std": 0.07624842971563339, "rewards/judge_quality/mean": 0.26749998331069946, "rewards/judge_quality/std": 0.10375107079744339, "rewards/total_composite/mean": 0.5953642725944519, "rewards/total_composite/std": 0.17409548163414001, "reward": 0.5953642725944519, "reward_std": 0.17409546673297882, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12591183185577393, "sampling/sampling_logp_difference/max": 2.2909507751464844, "sampling/importance_sampling_ratio/min": 0.10117022693157196, "sampling/importance_sampling_ratio/mean": 1.0110290050506592, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.745012141764164, "clip_ratio/low_mean": 0.06060696579515934, "clip_ratio/low_min": 0.06060696579515934, "clip_ratio/high_mean": 0.061670136637985706, "clip_ratio/high_max": 0.061670136637985706, "clip_ratio/region_mean": 0.12227710243314505, "reward_total_mean": 0.5953642725944519, "reward_meter_mean": 0.6033499240875244, "reward_meter_std": 0.3766990900039673, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9360681176185608, "reward_repeat_soft_std": 0.07624842971563339, "reward_judge_quality_mean": 0.26749998331069946, "reward_judge_quality_std": 0.10375107079744339, "reward_total_composite_mean": 0.5953642725944519, "reward_total_composite_std": 0.17409548163414001} {"timestamp_utc": "2026-04-13T03:16:33Z", "mode": "train", "global_step": 1971, "epoch": 0.19799095931692617, "loss": 0.0358, "grad_norm": 10.00031852722168, "learning_rate": 4.030303030303031e-06, "num_tokens": 3602662.0, "completions/mean_length": 66.875, "completions/min_length": 59.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.875, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9895226955413818, "rewards/meter/std": 0.0027267225086688995, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9576811790466309, "rewards/repeat_soft/std": 0.03156834840774536, "rewards/judge_quality/mean": 0.48124998807907104, "rewards/judge_quality/std": 0.23503421247005463, "rewards/total_composite/mean": 0.8354283571243286, "rewards/total_composite/std": 0.07041526585817337, "reward": 0.8354283571243286, "reward_std": 0.07041526585817337, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10757820308208466, "sampling/sampling_logp_difference/max": 1.659533977508545, "sampling/importance_sampling_ratio/min": 0.19022761285305023, "sampling/importance_sampling_ratio/mean": 1.0088993310928345, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6075580976903439, "clip_ratio/low_mean": 0.05381924379616976, "clip_ratio/low_min": 0.05381924379616976, "clip_ratio/high_mean": 0.025741437450051308, "clip_ratio/high_max": 0.025741437450051308, "clip_ratio/region_mean": 0.07956068124622107, "reward_total_mean": 0.8354283571243286, "reward_meter_mean": 0.9895226955413818, "reward_meter_std": 0.0027267225086688995, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9576811790466309, "reward_repeat_soft_std": 0.03156834840774536, "reward_judge_quality_mean": 0.48124998807907104, "reward_judge_quality_std": 0.23503421247005463, "reward_total_composite_mean": 0.8354283571243286, "reward_total_composite_std": 0.07041526585817337} {"timestamp_utc": "2026-04-13T03:16:40Z", "mode": "train", "global_step": 1972, "epoch": 0.19809141135107985, "loss": 0.0136, "grad_norm": 5.151123046875, "learning_rate": 4.027272727272727e-06, "num_tokens": 3604722.0, "completions/mean_length": 76.5, "completions/min_length": 73.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.5, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.9922820329666138, "rewards/meter/std": 0.0015509749064221978, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8535909652709961, "rewards/repeat_soft/std": 0.07778802514076233, "rewards/judge_quality/mean": 0.3137499690055847, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7760109901428223, "rewards/total_composite/std": 0.02320384606719017, "reward": 0.7760109901428223, "reward_std": 0.023203851655125618, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04815073683857918, "sampling/sampling_logp_difference/max": 1.5567375421524048, "sampling/importance_sampling_ratio/min": 0.21082273125648499, "sampling/importance_sampling_ratio/mean": 1.002302885055542, "sampling/importance_sampling_ratio/max": 1.9017550945281982, "entropy": 0.2701207920908928, "clip_ratio/low_mean": 0.016335661290213466, "clip_ratio/low_min": 0.016335661290213466, "clip_ratio/high_mean": 0.016518471762537956, "clip_ratio/high_max": 0.016518471762537956, "clip_ratio/region_mean": 0.03285413305275142, "reward_total_mean": 0.7760109901428223, "reward_meter_mean": 0.9922820329666138, "reward_meter_std": 0.0015509749064221978, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8535909652709961, "reward_repeat_soft_std": 0.07778802514076233, "reward_judge_quality_mean": 0.3137499690055847, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7760109901428223, "reward_total_composite_std": 0.02320384606719017} {"timestamp_utc": "2026-04-13T03:16:46Z", "mode": "train", "global_step": 1973, "epoch": 0.19819186338523356, "loss": 0.0259, "grad_norm": 10.232416152954102, "learning_rate": 4.024242424242424e-06, "num_tokens": 3606485.0, "completions/mean_length": 60.375, "completions/min_length": 56.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.375, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9678995609283447, "rewards/meter/std": 0.05256415903568268, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9645400047302246, "rewards/repeat_soft/std": 0.03894336149096489, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8102588653564453, "rewards/total_composite/std": 0.024898668751120567, "reward": 0.8102588653564453, "reward_std": 0.024898666888475418, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0841277465224266, "sampling/sampling_logp_difference/max": 1.357801914215088, "sampling/importance_sampling_ratio/min": 0.25722557306289673, "sampling/importance_sampling_ratio/mean": 1.0130577087402344, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5249739736318588, "clip_ratio/low_mean": 0.007936508394777775, "clip_ratio/low_min": 0.007936508394777775, "clip_ratio/high_mean": 0.08553431555628777, "clip_ratio/high_max": 0.08553431555628777, "clip_ratio/region_mean": 0.09347082395106554, "reward_total_mean": 0.8102588653564453, "reward_meter_mean": 0.9678995609283447, "reward_meter_std": 0.05256415903568268, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9645400047302246, "reward_repeat_soft_std": 0.03894336149096489, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8102588653564453, "reward_total_composite_std": 0.024898668751120567} {"timestamp_utc": "2026-04-13T03:16:52Z", "mode": "train", "global_step": 1974, "epoch": 0.19829231541938724, "loss": 0.0653, "grad_norm": 8.926578521728516, "learning_rate": 4.0212121212121216e-06, "num_tokens": 3608114.0, "completions/mean_length": 56.625, "completions/min_length": 46.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.625, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9753042459487915, "rewards/meter/std": 0.027937479317188263, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9737225770950317, "rewards/repeat_soft/std": 0.020310280844569206, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.19255799055099487, "rewards/total_composite/mean": 0.8280091881752014, "rewards/total_composite/std": 0.05948888510465622, "reward": 0.8280091881752014, "reward_std": 0.059488896280527115, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12462899088859558, "sampling/sampling_logp_difference/max": 2.2291007041931152, "sampling/importance_sampling_ratio/min": 0.10762517154216766, "sampling/importance_sampling_ratio/mean": 1.0086073875427246, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7800166085362434, "clip_ratio/low_mean": 0.09491996560245752, "clip_ratio/low_min": 0.09491996560245752, "clip_ratio/high_mean": 0.016304347664117813, "clip_ratio/high_max": 0.016304347664117813, "clip_ratio/region_mean": 0.11122431326657534, "reward_total_mean": 0.8280091881752014, "reward_meter_mean": 0.9753042459487915, "reward_meter_std": 0.027937479317188263, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9737225770950317, "reward_repeat_soft_std": 0.020310280844569206, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.19255799055099487, "reward_total_composite_mean": 0.8280091881752014, "reward_total_composite_std": 0.05948888510465622} {"timestamp_utc": "2026-04-13T03:17:00Z", "mode": "train", "global_step": 1975, "epoch": 0.19839276745354092, "loss": 0.0834, "grad_norm": 8.625997543334961, "learning_rate": 4.018181818181818e-06, "num_tokens": 3610109.0, "completions/mean_length": 79.375, "completions/min_length": 64.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.375, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.9690592885017395, "rewards/meter/std": 0.03292509913444519, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8824116587638855, "rewards/repeat_soft/std": 0.10753631591796875, "rewards/judge_quality/mean": 0.2762500047683716, "rewards/judge_quality/std": 0.12603145837783813, "rewards/total_composite/mean": 0.757192850112915, "rewards/total_composite/std": 0.050789974629879, "reward": 0.757192850112915, "reward_std": 0.0507899709045887, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10139559209346771, "sampling/sampling_logp_difference/max": 1.0570168495178223, "sampling/importance_sampling_ratio/min": 0.3474908769130707, "sampling/importance_sampling_ratio/mean": 1.006873607635498, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7002176493406296, "clip_ratio/low_mean": 0.03274922561831772, "clip_ratio/low_min": 0.03274922561831772, "clip_ratio/high_mean": 0.06159961502999067, "clip_ratio/high_max": 0.06159961502999067, "clip_ratio/region_mean": 0.0943488406483084, "reward_total_mean": 0.757192850112915, "reward_meter_mean": 0.9690592885017395, "reward_meter_std": 0.03292509913444519, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8824116587638855, "reward_repeat_soft_std": 0.10753631591796875, "reward_judge_quality_mean": 0.2762500047683716, "reward_judge_quality_std": 0.12603145837783813, "reward_total_composite_mean": 0.757192850112915, "reward_total_composite_std": 0.050789974629879} {"timestamp_utc": "2026-04-13T03:17:07Z", "mode": "train", "global_step": 1976, "epoch": 0.19849321948769463, "loss": 0.0535, "grad_norm": 9.224588394165039, "learning_rate": 4.015151515151515e-06, "num_tokens": 3612082.0, "completions/mean_length": 73.625, "completions/min_length": 66.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.625, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.9920152425765991, "rewards/meter/std": 0.003363202791661024, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.896185040473938, "rewards/repeat_soft/std": 0.05812975391745567, "rewards/judge_quality/mean": 0.30124998092651367, "rewards/judge_quality/std": 0.10398317128419876, "rewards/total_composite/mean": 0.7764003872871399, "rewards/total_composite/std": 0.03547009825706482, "reward": 0.7764003872871399, "reward_std": 0.035470105707645416, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10539549589157104, "sampling/sampling_logp_difference/max": 1.81358003616333, "sampling/importance_sampling_ratio/min": 0.16306930780410767, "sampling/importance_sampling_ratio/mean": 0.9972519874572754, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.42520896717906, "clip_ratio/low_mean": 0.040654988726601005, "clip_ratio/low_min": 0.040654988726601005, "clip_ratio/high_mean": 0.03521703602746129, "clip_ratio/high_max": 0.03521703602746129, "clip_ratio/region_mean": 0.0758720247540623, "reward_total_mean": 0.7764003872871399, "reward_meter_mean": 0.9920152425765991, "reward_meter_std": 0.003363202791661024, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.896185040473938, "reward_repeat_soft_std": 0.05812975391745567, "reward_judge_quality_mean": 0.30124998092651367, "reward_judge_quality_std": 0.10398317128419876, "reward_total_composite_mean": 0.7764003872871399, "reward_total_composite_std": 0.03547009825706482} {"timestamp_utc": "2026-04-13T03:17:14Z", "mode": "train", "global_step": 1977, "epoch": 0.19859367152184831, "loss": 0.0301, "grad_norm": 8.39686393737793, "learning_rate": 4.0121212121212125e-06, "num_tokens": 3614799.0, "completions/mean_length": 161.625, "completions/min_length": 144.0, "completions/max_length": 176.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 161.625, "completions/min_terminated_length": 144.0, "completions/max_terminated_length": 176.0, "rewards/meter/mean": 0.9231986999511719, "rewards/meter/std": 0.18571239709854126, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8796319961547852, "rewards/repeat_soft/std": 0.06967698782682419, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.0975411981344223, "rewards/total_composite/mean": 0.7666526436805725, "rewards/total_composite/std": 0.08224843442440033, "reward": 0.7666526436805725, "reward_std": 0.08224844187498093, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08602627366781235, "sampling/sampling_logp_difference/max": 2.0828568935394287, "sampling/importance_sampling_ratio/min": 0.12457380443811417, "sampling/importance_sampling_ratio/mean": 1.0031343698501587, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4114192724227905, "clip_ratio/low_mean": 0.017629834823310375, "clip_ratio/low_min": 0.017629834823310375, "clip_ratio/high_mean": 0.07001609494909644, "clip_ratio/high_max": 0.07001609494909644, "clip_ratio/region_mean": 0.08764592977240682, "reward_total_mean": 0.7666526436805725, "reward_meter_mean": 0.9231986999511719, "reward_meter_std": 0.18571239709854126, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8796319961547852, "reward_repeat_soft_std": 0.06967698782682419, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.0975411981344223, "reward_total_composite_mean": 0.7666526436805725, "reward_total_composite_std": 0.08224843442440033} {"timestamp_utc": "2026-04-13T03:17:21Z", "mode": "train", "global_step": 1978, "epoch": 0.19869412355600202, "loss": 0.1055, "grad_norm": 9.978693008422852, "learning_rate": 4.009090909090909e-06, "num_tokens": 3616575.0, "completions/mean_length": 62.0, "completions/min_length": 53.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9276936054229736, "rewards/meter/std": 0.018743213266134262, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9082189798355103, "rewards/repeat_soft/std": 0.04893266782164574, "rewards/judge_quality/mean": 0.3137499690055847, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7524089813232422, "rewards/total_composite/std": 0.03439759835600853, "reward": 0.7524089813232422, "reward_std": 0.03439760208129883, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0639992207288742, "sampling/sampling_logp_difference/max": 1.162731647491455, "sampling/importance_sampling_ratio/min": 0.3126310110092163, "sampling/importance_sampling_ratio/mean": 1.0125178098678589, "sampling/importance_sampling_ratio/max": 1.787527322769165, "entropy": 0.31739880703389645, "clip_ratio/low_mean": 0.027225365513004363, "clip_ratio/low_min": 0.027225365513004363, "clip_ratio/high_mean": 0.01549284323118627, "clip_ratio/high_max": 0.01549284323118627, "clip_ratio/region_mean": 0.04271820874419063, "reward_total_mean": 0.7524089813232422, "reward_meter_mean": 0.9276936054229736, "reward_meter_std": 0.018743213266134262, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9082189798355103, "reward_repeat_soft_std": 0.04893266782164574, "reward_judge_quality_mean": 0.3137499690055847, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7524089813232422, "reward_total_composite_std": 0.03439759835600853} {"timestamp_utc": "2026-04-13T03:17:27Z", "mode": "train", "global_step": 1979, "epoch": 0.1987945755901557, "loss": 0.1119, "grad_norm": 9.618964195251465, "learning_rate": 4.006060606060607e-06, "num_tokens": 3618225.0, "completions/mean_length": 62.25, "completions/min_length": 50.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.25, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9610710740089417, "rewards/meter/std": 0.043895572423934937, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9831219911575317, "rewards/repeat_soft/std": 0.0442165844142437, "rewards/judge_quality/mean": 0.44875001907348633, "rewards/judge_quality/std": 0.21256513893604279, "rewards/total_composite/mean": 0.8154191970825195, "rewards/total_composite/std": 0.05766848102211952, "reward": 0.8154191970825195, "reward_std": 0.05766846612095833, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11812078952789307, "sampling/sampling_logp_difference/max": 3.543675184249878, "sampling/importance_sampling_ratio/min": 0.028906894847750664, "sampling/importance_sampling_ratio/mean": 0.9988829493522644, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5673338584601879, "clip_ratio/low_mean": 0.04114624485373497, "clip_ratio/low_min": 0.04114624485373497, "clip_ratio/high_mean": 0.08604418579488993, "clip_ratio/high_max": 0.08604418579488993, "clip_ratio/region_mean": 0.1271904306486249, "reward_total_mean": 0.8154191970825195, "reward_meter_mean": 0.9610710740089417, "reward_meter_std": 0.043895572423934937, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9831219911575317, "reward_repeat_soft_std": 0.0442165844142437, "reward_judge_quality_mean": 0.44875001907348633, "reward_judge_quality_std": 0.21256513893604279, "reward_total_composite_mean": 0.8154191970825195, "reward_total_composite_std": 0.05766848102211952} {"timestamp_utc": "2026-04-13T03:17:33Z", "mode": "train", "global_step": 1980, "epoch": 0.19889502762430938, "loss": 0.034, "grad_norm": 6.95891809463501, "learning_rate": 4.003030303030303e-06, "num_tokens": 3620006.0, "completions/mean_length": 67.625, "completions/min_length": 63.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.625, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9749587774276733, "rewards/meter/std": 0.007890216074883938, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9862224459648132, "rewards/repeat_soft/std": 0.013683516532182693, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.806978702545166, "rewards/total_composite/std": 0.018253494054079056, "reward": 0.806978702545166, "reward_std": 0.018253494054079056, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11633963882923126, "sampling/sampling_logp_difference/max": 2.4511022567749023, "sampling/importance_sampling_ratio/min": 0.08619852364063263, "sampling/importance_sampling_ratio/mean": 1.0054219961166382, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5839112438261509, "clip_ratio/low_mean": 0.019874585792422295, "clip_ratio/low_min": 0.019874585792422295, "clip_ratio/high_mean": 0.07664647419005632, "clip_ratio/high_max": 0.07664647419005632, "clip_ratio/region_mean": 0.09652105998247862, "reward_total_mean": 0.806978702545166, "reward_meter_mean": 0.9749587774276733, "reward_meter_std": 0.007890216074883938, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9862224459648132, "reward_repeat_soft_std": 0.013683516532182693, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.806978702545166, "reward_total_composite_std": 0.018253494054079056} {"timestamp_utc": "2026-04-13T03:17:40Z", "mode": "train", "global_step": 1981, "epoch": 0.1989954796584631, "loss": -0.0778, "grad_norm": 11.548665046691895, "learning_rate": 4.000000000000001e-06, "num_tokens": 3621888.0, "completions/mean_length": 67.25, "completions/min_length": 38.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.25, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9835361242294312, "rewards/meter/std": 0.004354878328740597, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9646803140640259, "rewards/repeat_soft/std": 0.024126440286636353, "rewards/judge_quality/mean": 0.42624998092651367, "rewards/judge_quality/std": 0.21980105340480804, "rewards/total_composite/mean": 0.8075593113899231, "rewards/total_composite/std": 0.08008354157209396, "reward": 0.8075593113899231, "reward_std": 0.08008351922035217, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11264175176620483, "sampling/sampling_logp_difference/max": 1.5012211799621582, "sampling/importance_sampling_ratio/min": 0.27076345682144165, "sampling/importance_sampling_ratio/mean": 1.0047645568847656, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.620374970138073, "clip_ratio/low_mean": 0.031547619961202145, "clip_ratio/low_min": 0.031547619961202145, "clip_ratio/high_mean": 0.08029672224074602, "clip_ratio/high_max": 0.08029672224074602, "clip_ratio/region_mean": 0.11184434220194817, "reward_total_mean": 0.8075593113899231, "reward_meter_mean": 0.9835361242294312, "reward_meter_std": 0.004354878328740597, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9646803140640259, "reward_repeat_soft_std": 0.024126440286636353, "reward_judge_quality_mean": 0.42624998092651367, "reward_judge_quality_std": 0.21980105340480804, "reward_total_composite_mean": 0.8075593113899231, "reward_total_composite_std": 0.08008354157209396} {"timestamp_utc": "2026-04-13T03:17:46Z", "mode": "train", "global_step": 1982, "epoch": 0.19909593169261677, "loss": 0.0333, "grad_norm": 12.887622833251953, "learning_rate": 3.996969696969698e-06, "num_tokens": 3623510.0, "completions/mean_length": 40.75, "completions/min_length": 33.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.5439114570617676, "rewards/meter/std": 0.4498307704925537, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9567466974258423, "rewards/repeat_soft/std": 0.056118935346603394, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404937028885, "rewards/total_composite/mean": 0.6573098301887512, "rewards/total_composite/std": 0.2515038549900055, "reward": 0.6573098301887512, "reward_std": 0.2515038847923279, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11364232003688812, "sampling/sampling_logp_difference/max": 1.4053099155426025, "sampling/importance_sampling_ratio/min": 0.2742122411727905, "sampling/importance_sampling_ratio/mean": 1.0125545263290405, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5198860205709934, "clip_ratio/low_mean": 0.05952050909399986, "clip_ratio/low_min": 0.05952050909399986, "clip_ratio/high_mean": 0.05912503506988287, "clip_ratio/high_max": 0.05912503506988287, "clip_ratio/region_mean": 0.11864554416388273, "reward_total_mean": 0.6573098301887512, "reward_meter_mean": 0.5439114570617676, "reward_meter_std": 0.4498307704925537, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9567466974258423, "reward_repeat_soft_std": 0.056118935346603394, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404937028885, "reward_total_composite_mean": 0.6573098301887512, "reward_total_composite_std": 0.2515038549900055} {"timestamp_utc": "2026-04-13T03:17:53Z", "mode": "train", "global_step": 1983, "epoch": 0.19919638372677045, "loss": 0.0129, "grad_norm": 5.544788837432861, "learning_rate": 3.993939393939394e-06, "num_tokens": 3626356.0, "completions/mean_length": 157.75, "completions/min_length": 140.0, "completions/max_length": 169.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 157.75, "completions/min_terminated_length": 140.0, "completions/max_terminated_length": 169.0, "rewards/meter/mean": 0.9049679040908813, "rewards/meter/std": 0.08563661575317383, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9336860179901123, "rewards/repeat_soft/std": 0.021523868665099144, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.7499791383743286, "rewards/total_composite/std": 0.046838775277137756, "reward": 0.7499791383743286, "reward_std": 0.04683877155184746, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09006471186876297, "sampling/sampling_logp_difference/max": 1.6287050247192383, "sampling/importance_sampling_ratio/min": 0.196183443069458, "sampling/importance_sampling_ratio/mean": 1.0075334310531616, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4917479082942009, "clip_ratio/low_mean": 0.038423702120780945, "clip_ratio/low_min": 0.038423702120780945, "clip_ratio/high_mean": 0.04021429689601064, "clip_ratio/high_max": 0.04021429689601064, "clip_ratio/region_mean": 0.07863799901679158, "reward_total_mean": 0.7499791383743286, "reward_meter_mean": 0.9049679040908813, "reward_meter_std": 0.08563661575317383, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9336860179901123, "reward_repeat_soft_std": 0.021523868665099144, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.7499791383743286, "reward_total_composite_std": 0.046838775277137756} {"timestamp_utc": "2026-04-13T03:18:00Z", "mode": "train", "global_step": 1984, "epoch": 0.19929683576092416, "loss": 0.0882, "grad_norm": 6.874968528747559, "learning_rate": 3.990909090909092e-06, "num_tokens": 3628325.0, "completions/mean_length": 71.125, "completions/min_length": 55.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.125, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.9817807674407959, "rewards/meter/std": 0.021597350016236305, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8352291584014893, "rewards/repeat_soft/std": 0.14132148027420044, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.8388242721557617, "rewards/total_composite/std": 0.06866965442895889, "reward": 0.8388242721557617, "reward_std": 0.0686696469783783, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07606831192970276, "sampling/sampling_logp_difference/max": 1.4414608478546143, "sampling/importance_sampling_ratio/min": 0.23658190667629242, "sampling/importance_sampling_ratio/mean": 1.0138791799545288, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39498254656791687, "clip_ratio/low_mean": 0.06616742815822363, "clip_ratio/low_min": 0.06616742815822363, "clip_ratio/high_mean": 0.017847593873739243, "clip_ratio/high_max": 0.017847593873739243, "clip_ratio/region_mean": 0.08401502203196287, "reward_total_mean": 0.8388242721557617, "reward_meter_mean": 0.9817807674407959, "reward_meter_std": 0.021597350016236305, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8352291584014893, "reward_repeat_soft_std": 0.14132148027420044, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.8388242721557617, "reward_total_composite_std": 0.06866965442895889} {"timestamp_utc": "2026-04-13T03:18:06Z", "mode": "train", "global_step": 1985, "epoch": 0.19939728779507784, "loss": 0.048, "grad_norm": 9.728296279907227, "learning_rate": 3.987878787878788e-06, "num_tokens": 3630188.0, "completions/mean_length": 70.875, "completions/min_length": 59.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.875, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.8100948333740234, "rewards/meter/std": 0.2931669056415558, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9614613056182861, "rewards/repeat_soft/std": 0.024891328066587448, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.7201887369155884, "rewards/total_composite/std": 0.1537574976682663, "reward": 0.7201887369155884, "reward_std": 0.1537574827671051, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11604411154985428, "sampling/sampling_logp_difference/max": 1.5078084468841553, "sampling/importance_sampling_ratio/min": 0.22139465808868408, "sampling/importance_sampling_ratio/mean": 1.0130853652954102, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7078904807567596, "clip_ratio/low_mean": 0.022186147514730692, "clip_ratio/low_min": 0.022186147514730692, "clip_ratio/high_mean": 0.10444438271224499, "clip_ratio/high_max": 0.10444438271224499, "clip_ratio/region_mean": 0.12663053022697568, "reward_total_mean": 0.7201887369155884, "reward_meter_mean": 0.8100948333740234, "reward_meter_std": 0.2931669056415558, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9614613056182861, "reward_repeat_soft_std": 0.024891328066587448, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.7201887369155884, "reward_total_composite_std": 0.1537574976682663} {"timestamp_utc": "2026-04-13T03:18:13Z", "mode": "train", "global_step": 1986, "epoch": 0.19949773982923155, "loss": 0.0484, "grad_norm": 9.65299129486084, "learning_rate": 3.984848484848485e-06, "num_tokens": 3632023.0, "completions/mean_length": 55.375, "completions/min_length": 48.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.375, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9215386509895325, "rewards/meter/std": 0.15020494163036346, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9657369256019592, "rewards/repeat_soft/std": 0.030991191044449806, "rewards/judge_quality/mean": 0.5612499713897705, "rewards/judge_quality/std": 0.25614938139915466, "rewards/total_composite/mean": 0.8296411037445068, "rewards/total_composite/std": 0.11700858920812607, "reward": 0.8296411037445068, "reward_std": 0.11700858175754547, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11592625081539154, "sampling/sampling_logp_difference/max": 2.0521292686462402, "sampling/importance_sampling_ratio/min": 0.1284610778093338, "sampling/importance_sampling_ratio/mean": 0.9758694171905518, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.437993161380291, "clip_ratio/low_mean": 0.046899192966520786, "clip_ratio/low_min": 0.046899192966520786, "clip_ratio/high_mean": 0.03781791124492884, "clip_ratio/high_max": 0.03781791124492884, "clip_ratio/region_mean": 0.08471710421144962, "reward_total_mean": 0.8296411037445068, "reward_meter_mean": 0.9215386509895325, "reward_meter_std": 0.15020494163036346, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9657369256019592, "reward_repeat_soft_std": 0.030991191044449806, "reward_judge_quality_mean": 0.5612499713897705, "reward_judge_quality_std": 0.25614938139915466, "reward_total_composite_mean": 0.8296411037445068, "reward_total_composite_std": 0.11700858920812607} {"timestamp_utc": "2026-04-13T03:18:20Z", "mode": "train", "global_step": 1987, "epoch": 0.19959819186338523, "loss": 0.0009, "grad_norm": 8.872049331665039, "learning_rate": 3.9818181818181825e-06, "num_tokens": 3633966.0, "completions/mean_length": 79.875, "completions/min_length": 72.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.875, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.8519625663757324, "rewards/meter/std": 0.25476527214050293, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9082247018814087, "rewards/repeat_soft/std": 0.1033540666103363, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.12631450593471527, "rewards/total_composite/mean": 0.7235805988311768, "rewards/total_composite/std": 0.10475075244903564, "reward": 0.7235805988311768, "reward_std": 0.10475073754787445, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10452692955732346, "sampling/sampling_logp_difference/max": 2.286329746246338, "sampling/importance_sampling_ratio/min": 0.10163882374763489, "sampling/importance_sampling_ratio/mean": 0.9968690276145935, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5753087922930717, "clip_ratio/low_mean": 0.04851829633116722, "clip_ratio/low_min": 0.04851829633116722, "clip_ratio/high_mean": 0.04878252651542425, "clip_ratio/high_max": 0.04878252651542425, "clip_ratio/region_mean": 0.09730082284659147, "reward_total_mean": 0.7235805988311768, "reward_meter_mean": 0.8519625663757324, "reward_meter_std": 0.25476527214050293, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9082247018814087, "reward_repeat_soft_std": 0.1033540666103363, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.12631450593471527, "reward_total_composite_mean": 0.7235805988311768, "reward_total_composite_std": 0.10475075244903564} {"timestamp_utc": "2026-04-13T03:18:27Z", "mode": "train", "global_step": 1988, "epoch": 0.1996986438975389, "loss": 0.041, "grad_norm": 5.7051520347595215, "learning_rate": 3.978787878787879e-06, "num_tokens": 3636894.0, "completions/mean_length": 168.0, "completions/min_length": 163.0, "completions/max_length": 185.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 168.0, "completions/min_terminated_length": 163.0, "completions/max_terminated_length": 185.0, "rewards/meter/mean": 0.8683179020881653, "rewards/meter/std": 0.21306270360946655, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8703595399856567, "rewards/repeat_soft/std": 0.06406455487012863, "rewards/judge_quality/mean": 0.30124998092651367, "rewards/judge_quality/std": 0.10398317128419876, "rewards/total_composite/mean": 0.7181540131568909, "rewards/total_composite/std": 0.1023092195391655, "reward": 0.7181540131568909, "reward_std": 0.1023092046380043, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09016739577054977, "sampling/sampling_logp_difference/max": 1.5545687675476074, "sampling/importance_sampling_ratio/min": 0.2112804651260376, "sampling/importance_sampling_ratio/mean": 1.0098655223846436, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.517855066806078, "clip_ratio/low_mean": 0.015903098043054342, "clip_ratio/low_min": 0.015903098043054342, "clip_ratio/high_mean": 0.06870137434452772, "clip_ratio/high_max": 0.06870137434452772, "clip_ratio/region_mean": 0.08460447238758206, "reward_total_mean": 0.7181540131568909, "reward_meter_mean": 0.8683179020881653, "reward_meter_std": 0.21306270360946655, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8703595399856567, "reward_repeat_soft_std": 0.06406455487012863, "reward_judge_quality_mean": 0.30124998092651367, "reward_judge_quality_std": 0.10398317128419876, "reward_total_composite_mean": 0.7181540131568909, "reward_total_composite_std": 0.1023092195391655} {"timestamp_utc": "2026-04-13T03:18:39Z", "mode": "train", "global_step": 1989, "epoch": 0.19979909593169262, "loss": -0.2112, "grad_norm": 2.59317684173584, "learning_rate": 3.975757575757576e-06, "num_tokens": 3639290.0, "completions/mean_length": 186.5, "completions/min_length": 129.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 140.0, "completions/min_terminated_length": 129.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.7979443073272705, "rewards/meter/std": 0.22472934424877167, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9452509880065918, "rewards/repeat_soft/std": 0.025552857667207718, "rewards/judge_quality/mean": 0.26375001668930054, "rewards/judge_quality/std": 0.14401760697364807, "rewards/total_composite/mean": 0.5949071645736694, "rewards/total_composite/std": 0.26416274905204773, "reward": 0.5949071645736694, "reward_std": 0.26416274905204773, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12905441224575043, "sampling/sampling_logp_difference/max": 2.3697712421417236, "sampling/importance_sampling_ratio/min": 0.09350211173295975, "sampling/importance_sampling_ratio/mean": 1.0041135549545288, "sampling/importance_sampling_ratio/max": 1.938798427581787, "entropy": 0.6814002469182014, "clip_ratio/low_mean": 0.022113787941634655, "clip_ratio/low_min": 0.022113787941634655, "clip_ratio/high_mean": 0.0680027473717928, "clip_ratio/high_max": 0.0680027473717928, "clip_ratio/region_mean": 0.09011653531342745, "reward_total_mean": 0.5949071645736694, "reward_meter_mean": 0.7979443073272705, "reward_meter_std": 0.22472934424877167, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.2651650309562683, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9452509880065918, "reward_repeat_soft_std": 0.025552857667207718, "reward_judge_quality_mean": 0.26375001668930054, "reward_judge_quality_std": 0.14401760697364807, "reward_total_composite_mean": 0.5949071645736694, "reward_total_composite_std": 0.26416274905204773} {"timestamp_utc": "2026-04-13T03:18:46Z", "mode": "train", "global_step": 1990, "epoch": 0.1998995479658463, "loss": 0.0058, "grad_norm": 7.529413223266602, "learning_rate": 3.972727272727273e-06, "num_tokens": 3641690.0, "completions/mean_length": 122.0, "completions/min_length": 117.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.0, "completions/min_terminated_length": 117.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.807511568069458, "rewards/meter/std": 0.24247996509075165, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9100059270858765, "rewards/repeat_soft/std": 0.03473174571990967, "rewards/judge_quality/mean": 0.39375001192092896, "rewards/judge_quality/std": 0.15638209879398346, "rewards/total_composite/mean": 0.7225058078765869, "rewards/total_composite/std": 0.14055119454860687, "reward": 0.7225058078765869, "reward_std": 0.14055119454860687, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10921884328126907, "sampling/sampling_logp_difference/max": 2.8269214630126953, "sampling/importance_sampling_ratio/min": 0.05919480696320534, "sampling/importance_sampling_ratio/mean": 0.9992831349372864, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5063177794218063, "clip_ratio/low_mean": 0.021771694533526897, "clip_ratio/low_min": 0.021771694533526897, "clip_ratio/high_mean": 0.08020904008299112, "clip_ratio/high_max": 0.08020904008299112, "clip_ratio/region_mean": 0.10198073461651802, "reward_total_mean": 0.7225058078765869, "reward_meter_mean": 0.807511568069458, "reward_meter_std": 0.24247996509075165, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9100059270858765, "reward_repeat_soft_std": 0.03473174571990967, "reward_judge_quality_mean": 0.39375001192092896, "reward_judge_quality_std": 0.15638209879398346, "reward_total_composite_mean": 0.7225058078765869, "reward_total_composite_std": 0.14055119454860687} {"timestamp_utc": "2026-04-13T03:18:53Z", "mode": "train", "global_step": 1991, "epoch": 0.2, "loss": 0.0488, "grad_norm": 12.722031593322754, "learning_rate": 3.96969696969697e-06, "num_tokens": 3643410.0, "completions/mean_length": 55.0, "completions/min_length": 47.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.0, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.950374960899353, "rewards/meter/std": 0.04466971755027771, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8907173275947571, "rewards/repeat_soft/std": 0.0553828626871109, "rewards/judge_quality/mean": 0.44624999165534973, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7912404537200928, "rewards/total_composite/std": 0.041736021637916565, "reward": 0.7912404537200928, "reward_std": 0.04173601418733597, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1147942915558815, "sampling/sampling_logp_difference/max": 3.759897232055664, "sampling/importance_sampling_ratio/min": 0.023286134004592896, "sampling/importance_sampling_ratio/mean": 1.0201762914657593, "sampling/importance_sampling_ratio/max": 1.9543380737304688, "entropy": 0.6257264316082001, "clip_ratio/low_mean": 0.041724199429154396, "clip_ratio/low_min": 0.041724199429154396, "clip_ratio/high_mean": 0.06563364714384079, "clip_ratio/high_max": 0.06563364714384079, "clip_ratio/region_mean": 0.10735784657299519, "reward_total_mean": 0.7912404537200928, "reward_meter_mean": 0.950374960899353, "reward_meter_std": 0.04466971755027771, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8907173275947571, "reward_repeat_soft_std": 0.0553828626871109, "reward_judge_quality_mean": 0.44624999165534973, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7912404537200928, "reward_total_composite_std": 0.041736021637916565} {"timestamp_utc": "2026-04-13T03:18:59Z", "mode": "train", "global_step": 1992, "epoch": 0.2001004520341537, "loss": -0.0084, "grad_norm": 6.739052772521973, "learning_rate": 3.966666666666667e-06, "num_tokens": 3645574.0, "completions/mean_length": 107.5, "completions/min_length": 103.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.5, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.9082156419754028, "rewards/meter/std": 0.12200584262609482, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9365607500076294, "rewards/repeat_soft/std": 0.027335522696375847, "rewards/judge_quality/mean": 0.3187499940395355, "rewards/judge_quality/std": 0.1397382616996765, "rewards/total_composite/mean": 0.7432906627655029, "rewards/total_composite/std": 0.061635687947273254, "reward": 0.7432906627655029, "reward_std": 0.06163567304611206, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0943513885140419, "sampling/sampling_logp_difference/max": 1.5655345916748047, "sampling/importance_sampling_ratio/min": 0.20897626876831055, "sampling/importance_sampling_ratio/mean": 1.0103555917739868, "sampling/importance_sampling_ratio/max": 1.9079684019088745, "entropy": 0.5681710354983807, "clip_ratio/low_mean": 0.052122291177511215, "clip_ratio/low_min": 0.052122291177511215, "clip_ratio/high_mean": 0.029279429465532303, "clip_ratio/high_max": 0.029279429465532303, "clip_ratio/region_mean": 0.08140172064304352, "reward_total_mean": 0.7432906627655029, "reward_meter_mean": 0.9082156419754028, "reward_meter_std": 0.12200584262609482, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9365607500076294, "reward_repeat_soft_std": 0.027335522696375847, "reward_judge_quality_mean": 0.3187499940395355, "reward_judge_quality_std": 0.1397382616996765, "reward_total_composite_mean": 0.7432906627655029, "reward_total_composite_std": 0.061635687947273254} {"timestamp_utc": "2026-04-13T03:19:06Z", "mode": "train", "global_step": 1993, "epoch": 0.20020090406830737, "loss": 0.052, "grad_norm": 7.015955924987793, "learning_rate": 3.963636363636364e-06, "num_tokens": 3647664.0, "completions/mean_length": 94.25, "completions/min_length": 79.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.25, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.9779907464981079, "rewards/meter/std": 0.008126015774905682, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9364312887191772, "rewards/repeat_soft/std": 0.03977659344673157, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7906140089035034, "rewards/total_composite/std": 0.029771165922284126, "reward": 0.7906140089035034, "reward_std": 0.029771175235509872, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09194441884756088, "sampling/sampling_logp_difference/max": 1.6526203155517578, "sampling/importance_sampling_ratio/min": 0.19154733419418335, "sampling/importance_sampling_ratio/mean": 1.0069656372070312, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.522612739354372, "clip_ratio/low_mean": 0.02965856483206153, "clip_ratio/low_min": 0.02965856483206153, "clip_ratio/high_mean": 0.05615560431033373, "clip_ratio/high_max": 0.05615560431033373, "clip_ratio/region_mean": 0.08581416914239526, "reward_total_mean": 0.7906140089035034, "reward_meter_mean": 0.9779907464981079, "reward_meter_std": 0.008126015774905682, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9364312887191772, "reward_repeat_soft_std": 0.03977659344673157, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7906140089035034, "reward_total_composite_std": 0.029771165922284126} {"timestamp_utc": "2026-04-13T03:19:12Z", "mode": "train", "global_step": 1994, "epoch": 0.20030135610246108, "loss": -0.0468, "grad_norm": 12.910449028015137, "learning_rate": 3.960606060606061e-06, "num_tokens": 3649057.0, "completions/mean_length": 22.125, "completions/min_length": 18.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.125, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.6766160726547241, "rewards/meter/std": 0.26906320452690125, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9357137680053711, "rewards/repeat_soft/std": 0.040389057248830795, "rewards/judge_quality/mean": 0.6187499761581421, "rewards/judge_quality/std": 0.24976776540279388, "rewards/total_composite/mean": 0.733673632144928, "rewards/total_composite/std": 0.16414819657802582, "reward": 0.733673632144928, "reward_std": 0.16414819657802582, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08671826124191284, "sampling/sampling_logp_difference/max": 0.6327791213989258, "sampling/importance_sampling_ratio/min": 0.5311137437820435, "sampling/importance_sampling_ratio/mean": 1.0107444524765015, "sampling/importance_sampling_ratio/max": 1.4909231662750244, "entropy": 0.5848502591252327, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09164222981780767, "clip_ratio/high_max": 0.09164222981780767, "clip_ratio/region_mean": 0.09164222981780767, "reward_total_mean": 0.733673632144928, "reward_meter_mean": 0.6766160726547241, "reward_meter_std": 0.26906320452690125, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9357137680053711, "reward_repeat_soft_std": 0.040389057248830795, "reward_judge_quality_mean": 0.6187499761581421, "reward_judge_quality_std": 0.24976776540279388, "reward_total_composite_mean": 0.733673632144928, "reward_total_composite_std": 0.16414819657802582} {"timestamp_utc": "2026-04-13T03:19:18Z", "mode": "train", "global_step": 1995, "epoch": 0.20040180813661476, "loss": -0.0221, "grad_norm": 20.635892868041992, "learning_rate": 3.957575757575758e-06, "num_tokens": 3650702.0, "completions/mean_length": 44.625, "completions/min_length": 34.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.625, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.8210254907608032, "rewards/meter/std": 0.25025758147239685, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9767947196960449, "rewards/repeat_soft/std": 0.03238915652036667, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7337659597396851, "rewards/total_composite/std": 0.10858356952667236, "reward": 0.7337659597396851, "reward_std": 0.10858356952667236, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11536166071891785, "sampling/sampling_logp_difference/max": 1.942896842956543, "sampling/importance_sampling_ratio/min": 0.1432882696390152, "sampling/importance_sampling_ratio/mean": 1.007911205291748, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5204624831676483, "clip_ratio/low_mean": 0.04796110885217786, "clip_ratio/low_min": 0.04796110885217786, "clip_ratio/high_mean": 0.04383293678984046, "clip_ratio/high_max": 0.04383293678984046, "clip_ratio/region_mean": 0.09179404564201832, "reward_total_mean": 0.7337659597396851, "reward_meter_mean": 0.8210254907608032, "reward_meter_std": 0.25025758147239685, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9767947196960449, "reward_repeat_soft_std": 0.03238915652036667, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7337659597396851, "reward_total_composite_std": 0.10858356952667236} {"timestamp_utc": "2026-04-13T03:19:25Z", "mode": "train", "global_step": 1996, "epoch": 0.20050226017076847, "loss": -0.0331, "grad_norm": 7.011910915374756, "learning_rate": 3.954545454545454e-06, "num_tokens": 3653018.0, "completions/mean_length": 109.5, "completions/min_length": 93.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.5, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.8385644555091858, "rewards/meter/std": 0.19971193373203278, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9111506938934326, "rewards/repeat_soft/std": 0.05931096896529198, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7380940914154053, "rewards/total_composite/std": 0.09747294336557388, "reward": 0.7380940914154053, "reward_std": 0.09747294336557388, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11274231225252151, "sampling/sampling_logp_difference/max": 2.8563408851623535, "sampling/importance_sampling_ratio/min": 0.05747869610786438, "sampling/importance_sampling_ratio/mean": 1.0001252889633179, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5663504265248775, "clip_ratio/low_mean": 0.03583450196310878, "clip_ratio/low_min": 0.03583450196310878, "clip_ratio/high_mean": 0.07300856243818998, "clip_ratio/high_max": 0.07300856243818998, "clip_ratio/region_mean": 0.10884306440129876, "reward_total_mean": 0.7380940914154053, "reward_meter_mean": 0.8385644555091858, "reward_meter_std": 0.19971193373203278, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9111506938934326, "reward_repeat_soft_std": 0.05931096896529198, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7380940914154053, "reward_total_composite_std": 0.09747294336557388} {"timestamp_utc": "2026-04-13T03:19:32Z", "mode": "train", "global_step": 1997, "epoch": 0.20060271220492215, "loss": 0.0302, "grad_norm": 11.874625205993652, "learning_rate": 3.951515151515152e-06, "num_tokens": 3655030.0, "completions/mean_length": 65.5, "completions/min_length": 54.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.5, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9142535924911499, "rewards/meter/std": 0.08670271933078766, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9730360507965088, "rewards/repeat_soft/std": 0.018351148813962936, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.7794677019119263, "rewards/total_composite/std": 0.04387584701180458, "reward": 0.7794677019119263, "reward_std": 0.04387585073709488, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10314653813838959, "sampling/sampling_logp_difference/max": 2.1064586639404297, "sampling/importance_sampling_ratio/min": 0.12166807055473328, "sampling/importance_sampling_ratio/mean": 1.000941514968872, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5806053280830383, "clip_ratio/low_mean": 0.026748252101242542, "clip_ratio/low_min": 0.026748252101242542, "clip_ratio/high_mean": 0.08960710931569338, "clip_ratio/high_max": 0.08960710931569338, "clip_ratio/region_mean": 0.11635536141693592, "reward_total_mean": 0.7794677019119263, "reward_meter_mean": 0.9142535924911499, "reward_meter_std": 0.08670271933078766, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9730360507965088, "reward_repeat_soft_std": 0.018351148813962936, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.7794677019119263, "reward_total_composite_std": 0.04387584701180458} {"timestamp_utc": "2026-04-13T03:19:39Z", "mode": "train", "global_step": 1998, "epoch": 0.20070316423907583, "loss": 0.0321, "grad_norm": 5.125667572021484, "learning_rate": 3.948484848484849e-06, "num_tokens": 3657024.0, "completions/mean_length": 129.25, "completions/min_length": 113.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 129.25, "completions/min_terminated_length": 113.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.8930000066757202, "rewards/meter/std": 0.14797236025333405, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8417840003967285, "rewards/repeat_soft/std": 0.09764155745506287, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.7365283966064453, "rewards/total_composite/std": 0.08762817829847336, "reward": 0.7365283966064453, "reward_std": 0.08762818574905396, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08530699461698532, "sampling/sampling_logp_difference/max": 1.4531121253967285, "sampling/importance_sampling_ratio/min": 0.2338414192199707, "sampling/importance_sampling_ratio/mean": 1.0119152069091797, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6357042901217937, "clip_ratio/low_mean": 0.03292026184499264, "clip_ratio/low_min": 0.03292026184499264, "clip_ratio/high_mean": 0.03555574594065547, "clip_ratio/high_max": 0.03555574594065547, "clip_ratio/region_mean": 0.06847600778564811, "reward_total_mean": 0.7365283966064453, "reward_meter_mean": 0.8930000066757202, "reward_meter_std": 0.14797236025333405, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8417840003967285, "reward_repeat_soft_std": 0.09764155745506287, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.7365283966064453, "reward_total_composite_std": 0.08762817829847336} {"timestamp_utc": "2026-04-13T03:19:50Z", "mode": "train", "global_step": 1999, "epoch": 0.20080361627322954, "loss": -0.2269, "grad_norm": 1.7725650072097778, "learning_rate": 3.945454545454545e-06, "num_tokens": 3659435.0, "completions/mean_length": 188.375, "completions/min_length": 117.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 142.1428680419922, "completions/min_terminated_length": 117.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.9471408128738403, "rewards/meter/std": 0.05248282104730606, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9340166449546814, "rewards/repeat_soft/std": 0.03798726946115494, "rewards/judge_quality/mean": 0.22999998927116394, "rewards/judge_quality/std": 0.13341663777828217, "rewards/total_composite/mean": 0.651694655418396, "rewards/total_composite/std": 0.26485252380371094, "reward": 0.651694655418396, "reward_std": 0.26485249400138855, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1056489422917366, "sampling/sampling_logp_difference/max": 2.2467188835144043, "sampling/importance_sampling_ratio/min": 0.10574561357498169, "sampling/importance_sampling_ratio/mean": 1.0119562149047852, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5552256554365158, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09015338122844696, "clip_ratio/high_max": 0.09015338122844696, "clip_ratio/region_mean": 0.09015338122844696, "reward_total_mean": 0.651694655418396, "reward_meter_mean": 0.9471408128738403, "reward_meter_std": 0.05248282104730606, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9340166449546814, "reward_repeat_soft_std": 0.03798726946115494, "reward_judge_quality_mean": 0.22999998927116394, "reward_judge_quality_std": 0.13341663777828217, "reward_total_composite_mean": 0.651694655418396, "reward_total_composite_std": 0.26485252380371094} {"timestamp_utc": "2026-04-13T03:19:56Z", "mode": "train", "global_step": 2000, "epoch": 0.20090406830738322, "loss": -0.0153, "grad_norm": 11.908234596252441, "learning_rate": 3.942424242424243e-06, "num_tokens": 3660960.0, "completions/mean_length": 29.625, "completions/min_length": 28.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.625, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9805858731269836, "rewards/meter/std": 0.022649794816970825, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9609359502792358, "rewards/repeat_soft/std": 0.0034086306113749743, "rewards/judge_quality/mean": 0.4312499761581421, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8167322278022766, "rewards/total_composite/std": 0.009322555735707283, "reward": 0.8167322278022766, "reward_std": 0.009322557598352432, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10852807015180588, "sampling/sampling_logp_difference/max": 2.2988531589508057, "sampling/importance_sampling_ratio/min": 0.10037389397621155, "sampling/importance_sampling_ratio/mean": 1.0149544477462769, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5308308191597462, "clip_ratio/low_mean": 0.034388408064842224, "clip_ratio/low_min": 0.034388408064842224, "clip_ratio/high_mean": 0.0748917767778039, "clip_ratio/high_max": 0.0748917767778039, "clip_ratio/region_mean": 0.10928018484264612, "reward_total_mean": 0.8167322278022766, "reward_meter_mean": 0.9805858731269836, "reward_meter_std": 0.022649794816970825, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9609359502792358, "reward_repeat_soft_std": 0.0034086306113749743, "reward_judge_quality_mean": 0.4312499761581421, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8167322278022766, "reward_total_composite_std": 0.009322555735707283} {"timestamp_utc": "2026-04-13T03:20:46Z", "mode": "eval", "global_step": 2000, "epoch": 0.20090406830738322, "eval_loss": NaN, "eval_runtime": 50.0705, "eval_samples_per_second": 1.598, "eval_steps_per_second": 0.2, "eval_num_tokens": 3660960.0, "eval_completions/mean_length": 107.0, "eval_completions/min_length": 43.4, "eval_completions/max_length": 206.3, "eval_completions/clipped_ratio": 0.0125, "eval_completions/mean_terminated_length": 101.80892868041992, "eval_completions/min_terminated_length": 43.4, "eval_completions/max_terminated_length": 173.6, "eval_rewards/meter/mean": 0.8778757333755494, "eval_rewards/meter/std": 0.1795883210375905, "eval_rewards/count_adherence/mean": 0.9787500023841857, "eval_rewards/count_adherence/std": 0.04986142441630363, "eval_rewards/hard_gate/mean": 0.9625, "eval_rewards/hard_gate/std": 0.10606601536273956, "eval_rewards/repeat_soft/mean": 0.9268425881862641, "eval_rewards/repeat_soft/std": 0.051629329472780226, "eval_rewards/judge_quality/mean": 0.35962500274181364, "eval_rewards/judge_quality/std": 0.14045618698000908, "eval_rewards/total_composite/mean": 0.7200792253017425, "eval_rewards/total_composite/std": 0.16240096241235732, "eval_reward": 0.7200792253017425, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.05174119770526886, "eval_sampling/sampling_logp_difference/max": 0.9759568691253662, "eval_sampling/importance_sampling_ratio/min": 0.382222643494606, "eval_sampling/importance_sampling_ratio/mean": 1.0118723869323731, "eval_sampling/importance_sampling_ratio/max": 1.4194300174713135, "eval_entropy": 0.5535867154598236, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7200792253017425, "eval_reward_meter_mean": 0.8778757333755494, "eval_reward_meter_std": 0.1795883210375905, "eval_reward_count_adherence_mean": 0.9787500023841857, "eval_reward_count_adherence_std": 0.04986142441630363, "eval_reward_hard_gate_mean": 0.9625, "eval_reward_hard_gate_std": 0.10606601536273956, "eval_reward_repeat_soft_mean": 0.9268425881862641, "eval_reward_repeat_soft_std": 0.051629329472780226, "eval_reward_judge_quality_mean": 0.35962500274181364, "eval_reward_judge_quality_std": 0.14045618698000908, "eval_reward_total_composite_mean": 0.7200792253017425, "eval_reward_total_composite_std": 0.16240096241235732} {"timestamp_utc": "2026-04-13T03:21:01Z", "mode": "train", "global_step": 2001, "epoch": 0.20100452034153693, "loss": -0.1498, "grad_norm": 1.6282521486282349, "learning_rate": 3.93939393939394e-06, "num_tokens": 3662625.0, "completions/mean_length": 114.125, "completions/min_length": 50.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 57.28571701049805, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.981220006942749, "rewards/meter/std": 0.012669369578361511, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9759023785591125, "rewards/repeat_soft/std": 0.02595800720155239, "rewards/judge_quality/mean": 0.3787499964237213, "rewards/judge_quality/std": 0.1537565290927887, "rewards/total_composite/mean": 0.7165762782096863, "rewards/total_composite/std": 0.2896397113800049, "reward": 0.7165762782096863, "reward_std": 0.2896396815776825, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11908666789531708, "sampling/sampling_logp_difference/max": 1.5462853908538818, "sampling/importance_sampling_ratio/min": 0.2130378633737564, "sampling/importance_sampling_ratio/mean": 1.009313702583313, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6649883985519409, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09675345849245787, "clip_ratio/high_max": 0.09675345849245787, "clip_ratio/region_mean": 0.09675345849245787, "reward_total_mean": 0.7165762782096863, "reward_meter_mean": 0.981220006942749, "reward_meter_std": 0.012669369578361511, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9759023785591125, "reward_repeat_soft_std": 0.02595800720155239, "reward_judge_quality_mean": 0.3787499964237213, "reward_judge_quality_std": 0.1537565290927887, "reward_total_composite_mean": 0.7165762782096863, "reward_total_composite_std": 0.2896397113800049} {"timestamp_utc": "2026-04-13T03:21:07Z", "mode": "train", "global_step": 2002, "epoch": 0.2011049723756906, "loss": 0.0309, "grad_norm": 11.612317085266113, "learning_rate": 3.936363636363636e-06, "num_tokens": 3664663.0, "completions/mean_length": 92.75, "completions/min_length": 87.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.75, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.8952130079269409, "rewards/meter/std": 0.2318383902311325, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9779855608940125, "rewards/repeat_soft/std": 0.016955891624093056, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7702693939208984, "rewards/total_composite/std": 0.12281252443790436, "reward": 0.7702693939208984, "reward_std": 0.12281252443790436, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11735213547945023, "sampling/sampling_logp_difference/max": 1.471841812133789, "sampling/importance_sampling_ratio/min": 0.22950239479541779, "sampling/importance_sampling_ratio/mean": 1.0115312337875366, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7202528789639473, "clip_ratio/low_mean": 0.015789473429322243, "clip_ratio/low_min": 0.015789473429322243, "clip_ratio/high_mean": 0.10527842119336128, "clip_ratio/high_max": 0.10527842119336128, "clip_ratio/region_mean": 0.12106789462268353, "reward_total_mean": 0.7702693939208984, "reward_meter_mean": 0.8952130079269409, "reward_meter_std": 0.2318383902311325, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9779855608940125, "reward_repeat_soft_std": 0.016955891624093056, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7702693939208984, "reward_total_composite_std": 0.12281252443790436} {"timestamp_utc": "2026-04-13T03:21:14Z", "mode": "train", "global_step": 2003, "epoch": 0.2012054244098443, "loss": 0.026, "grad_norm": 7.990621566772461, "learning_rate": 3.9333333333333335e-06, "num_tokens": 3666402.0, "completions/mean_length": 66.375, "completions/min_length": 62.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.375, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9459450244903564, "rewards/meter/std": 0.04010935500264168, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9333119988441467, "rewards/repeat_soft/std": 0.034185007214546204, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7822564840316772, "rewards/total_composite/std": 0.0223619993776083, "reward": 0.7822564840316772, "reward_std": 0.022361995652318, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09846122562885284, "sampling/sampling_logp_difference/max": 2.9339816570281982, "sampling/importance_sampling_ratio/min": 0.1769888550043106, "sampling/importance_sampling_ratio/mean": 1.001736044883728, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.531191986054182, "clip_ratio/low_mean": 0.056383362505584955, "clip_ratio/low_min": 0.056383362505584955, "clip_ratio/high_mean": 0.035527801141142845, "clip_ratio/high_max": 0.035527801141142845, "clip_ratio/region_mean": 0.0919111636467278, "reward_total_mean": 0.7822564840316772, "reward_meter_mean": 0.9459450244903564, "reward_meter_std": 0.04010935500264168, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9333119988441467, "reward_repeat_soft_std": 0.034185007214546204, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7822564840316772, "reward_total_composite_std": 0.0223619993776083} {"timestamp_utc": "2026-04-13T03:21:20Z", "mode": "train", "global_step": 2004, "epoch": 0.201305876443998, "loss": 0.0439, "grad_norm": 14.607230186462402, "learning_rate": 3.930303030303031e-06, "num_tokens": 3667971.0, "completions/mean_length": 50.125, "completions/min_length": 44.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.125, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.8453206419944763, "rewards/meter/std": 0.2639203667640686, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.94151371717453, "rewards/repeat_soft/std": 0.04146759212017059, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.10392305999994278, "rewards/total_composite/mean": 0.7640456557273865, "rewards/total_composite/std": 0.12400192022323608, "reward": 0.7640456557273865, "reward_std": 0.12400191277265549, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10900726914405823, "sampling/sampling_logp_difference/max": 3.080923080444336, "sampling/importance_sampling_ratio/min": 0.045916855335235596, "sampling/importance_sampling_ratio/mean": 1.0201947689056396, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6501789204776287, "clip_ratio/low_mean": 0.024038461968302727, "clip_ratio/low_min": 0.024038461968302727, "clip_ratio/high_mean": 0.08252215478569269, "clip_ratio/high_max": 0.08252215478569269, "clip_ratio/region_mean": 0.10656061675399542, "reward_total_mean": 0.7640456557273865, "reward_meter_mean": 0.8453206419944763, "reward_meter_std": 0.2639203667640686, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.94151371717453, "reward_repeat_soft_std": 0.04146759212017059, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.10392305999994278, "reward_total_composite_mean": 0.7640456557273865, "reward_total_composite_std": 0.12400192022323608} {"timestamp_utc": "2026-04-13T03:21:28Z", "mode": "train", "global_step": 2005, "epoch": 0.20140632847815168, "loss": 0.0098, "grad_norm": 4.127713680267334, "learning_rate": 3.927272727272727e-06, "num_tokens": 3671120.0, "completions/mean_length": 206.625, "completions/min_length": 179.0, "completions/max_length": 236.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 206.625, "completions/min_terminated_length": 179.0, "completions/max_terminated_length": 236.0, "rewards/meter/mean": 0.9943808913230896, "rewards/meter/std": 0.0030051504727452993, "rewards/count_adherence/mean": 0.9166666269302368, "rewards/count_adherence/std": 0.0890870913863182, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8266896605491638, "rewards/repeat_soft/std": 0.055848438292741776, "rewards/judge_quality/mean": 0.1837500035762787, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.722765326499939, "rewards/total_composite/std": 0.0360754132270813, "reward": 0.722765326499939, "reward_std": 0.0360754020512104, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06814489513635635, "sampling/sampling_logp_difference/max": 1.3959379196166992, "sampling/importance_sampling_ratio/min": 0.2476007044315338, "sampling/importance_sampling_ratio/mean": 1.0118398666381836, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4847365692257881, "clip_ratio/low_mean": 0.03753823321312666, "clip_ratio/low_min": 0.03753823321312666, "clip_ratio/high_mean": 0.02581630041822791, "clip_ratio/high_max": 0.02581630041822791, "clip_ratio/region_mean": 0.06335453363135457, "reward_total_mean": 0.722765326499939, "reward_meter_mean": 0.9943808913230896, "reward_meter_std": 0.0030051504727452993, "reward_count_adherence_mean": 0.9166666269302368, "reward_count_adherence_std": 0.0890870913863182, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8266896605491638, "reward_repeat_soft_std": 0.055848438292741776, "reward_judge_quality_mean": 0.1837500035762787, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.722765326499939, "reward_total_composite_std": 0.0360754132270813} {"timestamp_utc": "2026-04-13T03:21:37Z", "mode": "train", "global_step": 2006, "epoch": 0.20150678051230536, "loss": -0.0088, "grad_norm": 4.379899024963379, "learning_rate": 3.9242424242424244e-06, "num_tokens": 3674137.0, "completions/mean_length": 186.125, "completions/min_length": 169.0, "completions/max_length": 207.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 186.125, "completions/min_terminated_length": 169.0, "completions/max_terminated_length": 207.0, "rewards/meter/mean": 0.9923309683799744, "rewards/meter/std": 0.0052482010796666145, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7996487617492676, "rewards/repeat_soft/std": 0.07541874796152115, "rewards/judge_quality/mean": 0.29249998927116394, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7605137825012207, "rewards/total_composite/std": 0.03298250958323479, "reward": 0.7605137825012207, "reward_std": 0.032982513308525085, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07477398216724396, "sampling/sampling_logp_difference/max": 1.8772683143615723, "sampling/importance_sampling_ratio/min": 0.15300750732421875, "sampling/importance_sampling_ratio/mean": 1.0026830434799194, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43365662172436714, "clip_ratio/low_mean": 0.03261650074273348, "clip_ratio/low_min": 0.03261650074273348, "clip_ratio/high_mean": 0.03223726712167263, "clip_ratio/high_max": 0.03223726712167263, "clip_ratio/region_mean": 0.06485376786440611, "reward_total_mean": 0.7605137825012207, "reward_meter_mean": 0.9923309683799744, "reward_meter_std": 0.0052482010796666145, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7996487617492676, "reward_repeat_soft_std": 0.07541874796152115, "reward_judge_quality_mean": 0.29249998927116394, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7605137825012207, "reward_total_composite_std": 0.03298250958323479} {"timestamp_utc": "2026-04-13T03:21:45Z", "mode": "train", "global_step": 2007, "epoch": 0.20160723254645907, "loss": -0.0465, "grad_norm": 4.885383129119873, "learning_rate": 3.921212121212122e-06, "num_tokens": 3676988.0, "completions/mean_length": 153.375, "completions/min_length": 132.0, "completions/max_length": 173.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 153.375, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 173.0, "rewards/meter/mean": 0.990125298500061, "rewards/meter/std": 0.004948662593960762, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.10690449178218842, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8590087890625, "rewards/repeat_soft/std": 0.024237534031271935, "rewards/judge_quality/mean": 0.23749999701976776, "rewards/judge_quality/std": 0.0353553369641304, "rewards/total_composite/mean": 0.7227072715759277, "rewards/total_composite/std": 0.01719127595424652, "reward": 0.7227072715759277, "reward_std": 0.017191298305988312, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05433980002999306, "sampling/sampling_logp_difference/max": 1.4934179782867432, "sampling/importance_sampling_ratio/min": 0.22460365295410156, "sampling/importance_sampling_ratio/mean": 1.0077139139175415, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3404519818723202, "clip_ratio/low_mean": 0.012836021836847067, "clip_ratio/low_min": 0.012836021836847067, "clip_ratio/high_mean": 0.039679511450231075, "clip_ratio/high_max": 0.039679511450231075, "clip_ratio/region_mean": 0.05251553328707814, "reward_total_mean": 0.7227072715759277, "reward_meter_mean": 0.990125298500061, "reward_meter_std": 0.004948662593960762, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.10690449178218842, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8590087890625, "reward_repeat_soft_std": 0.024237534031271935, "reward_judge_quality_mean": 0.23749999701976776, "reward_judge_quality_std": 0.0353553369641304, "reward_total_composite_mean": 0.7227072715759277, "reward_total_composite_std": 0.01719127595424652} {"timestamp_utc": "2026-04-13T03:21:51Z", "mode": "train", "global_step": 2008, "epoch": 0.20170768458061275, "loss": 0.1529, "grad_norm": 18.2756404876709, "learning_rate": 3.918181818181819e-06, "num_tokens": 3678939.0, "completions/mean_length": 60.875, "completions/min_length": 49.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.8950519561767578, "rewards/meter/std": 0.23442816734313965, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9746439456939697, "rewards/repeat_soft/std": 0.03780348598957062, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.7469877600669861, "rewards/total_composite/std": 0.11789257079362869, "reward": 0.7469877600669861, "reward_std": 0.11789257079362869, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12034186720848083, "sampling/sampling_logp_difference/max": 1.8032360076904297, "sampling/importance_sampling_ratio/min": 0.16476485133171082, "sampling/importance_sampling_ratio/mean": 1.0107636451721191, "sampling/importance_sampling_ratio/max": 1.8314083814620972, "entropy": 0.6719830110669136, "clip_ratio/low_mean": 0.03127705678343773, "clip_ratio/low_min": 0.03127705678343773, "clip_ratio/high_mean": 0.08451180625706911, "clip_ratio/high_max": 0.08451180625706911, "clip_ratio/region_mean": 0.11578886304050684, "reward_total_mean": 0.7469877600669861, "reward_meter_mean": 0.8950519561767578, "reward_meter_std": 0.23442816734313965, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9746439456939697, "reward_repeat_soft_std": 0.03780348598957062, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.7469877600669861, "reward_total_composite_std": 0.11789257079362869} {"timestamp_utc": "2026-04-13T03:21:58Z", "mode": "train", "global_step": 2009, "epoch": 0.20180813661476646, "loss": 0.0473, "grad_norm": 7.49652099609375, "learning_rate": 3.915151515151515e-06, "num_tokens": 3680901.0, "completions/mean_length": 72.25, "completions/min_length": 62.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.25, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.4411693215370178, "rewards/meter/std": 0.3839699923992157, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9080817103385925, "rewards/repeat_soft/std": 0.04069387540221214, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.5525844097137451, "rewards/total_composite/std": 0.17637114226818085, "reward": 0.5525844097137451, "reward_std": 0.17637114226818085, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10209175944328308, "sampling/sampling_logp_difference/max": 2.4130935668945312, "sampling/importance_sampling_ratio/min": 0.08953787386417389, "sampling/importance_sampling_ratio/mean": 1.0006070137023926, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5868635848164558, "clip_ratio/low_mean": 0.05356888985261321, "clip_ratio/low_min": 0.05356888985261321, "clip_ratio/high_mean": 0.035818048752844334, "clip_ratio/high_max": 0.035818048752844334, "clip_ratio/region_mean": 0.08938693860545754, "reward_total_mean": 0.5525844097137451, "reward_meter_mean": 0.4411693215370178, "reward_meter_std": 0.3839699923992157, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9080817103385925, "reward_repeat_soft_std": 0.04069387540221214, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.5525844097137451, "reward_total_composite_std": 0.17637114226818085} {"timestamp_utc": "2026-04-13T03:22:05Z", "mode": "train", "global_step": 2010, "epoch": 0.20190858864892014, "loss": -0.0028, "grad_norm": 9.982221603393555, "learning_rate": 3.912121212121213e-06, "num_tokens": 3683325.0, "completions/mean_length": 125.0, "completions/min_length": 109.0, "completions/max_length": 137.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.0, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.8891465663909912, "rewards/meter/std": 0.218770831823349, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9135360717773438, "rewards/repeat_soft/std": 0.025415610522031784, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.7419695854187012, "rewards/total_composite/std": 0.09069477021694183, "reward": 0.7419695854187012, "reward_std": 0.09069477766752243, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08812545984983444, "sampling/sampling_logp_difference/max": 1.7296600341796875, "sampling/importance_sampling_ratio/min": 0.1773446798324585, "sampling/importance_sampling_ratio/mean": 0.9973096251487732, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.41660548746585846, "clip_ratio/low_mean": 0.0052966102957725525, "clip_ratio/low_min": 0.0052966102957725525, "clip_ratio/high_mean": 0.07417296431958675, "clip_ratio/high_max": 0.07417296431958675, "clip_ratio/region_mean": 0.0794695746153593, "reward_total_mean": 0.7419695854187012, "reward_meter_mean": 0.8891465663909912, "reward_meter_std": 0.218770831823349, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9135360717773438, "reward_repeat_soft_std": 0.025415610522031784, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.7419695854187012, "reward_total_composite_std": 0.09069477021694183} {"timestamp_utc": "2026-04-13T03:22:11Z", "mode": "train", "global_step": 2011, "epoch": 0.20200904068307382, "loss": -0.0381, "grad_norm": 13.075427055358887, "learning_rate": 3.90909090909091e-06, "num_tokens": 3684844.0, "completions/mean_length": 33.875, "completions/min_length": 29.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.875, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.6974789500236511, "rewards/meter/std": 0.31530460715293884, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9598709940910339, "rewards/repeat_soft/std": 0.007435863837599754, "rewards/judge_quality/mean": 0.8237500190734863, "rewards/judge_quality/std": 0.2722361385822296, "rewards/total_composite/mean": 0.8069776296615601, "rewards/total_composite/std": 0.1362701654434204, "reward": 0.8069776296615601, "reward_std": 0.1362701654434204, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12219718098640442, "sampling/sampling_logp_difference/max": 1.7794809341430664, "sampling/importance_sampling_ratio/min": 0.16872571408748627, "sampling/importance_sampling_ratio/mean": 1.0065295696258545, "sampling/importance_sampling_ratio/max": 1.8662257194519043, "entropy": 0.7280573323369026, "clip_ratio/low_mean": 0.04693136387504637, "clip_ratio/low_min": 0.04693136387504637, "clip_ratio/high_mean": 0.049893164075911045, "clip_ratio/high_max": 0.049893164075911045, "clip_ratio/region_mean": 0.09682452795095742, "reward_total_mean": 0.8069776296615601, "reward_meter_mean": 0.6974789500236511, "reward_meter_std": 0.31530460715293884, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9598709940910339, "reward_repeat_soft_std": 0.007435863837599754, "reward_judge_quality_mean": 0.8237500190734863, "reward_judge_quality_std": 0.2722361385822296, "reward_total_composite_mean": 0.8069776296615601, "reward_total_composite_std": 0.1362701654434204} {"timestamp_utc": "2026-04-13T03:22:18Z", "mode": "train", "global_step": 2012, "epoch": 0.20210949271722753, "loss": 0.0139, "grad_norm": 9.194389343261719, "learning_rate": 3.906060606060606e-06, "num_tokens": 3686894.0, "completions/mean_length": 93.25, "completions/min_length": 74.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.25, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.5864174365997314, "rewards/meter/std": 0.2772681415081024, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8702847957611084, "rewards/repeat_soft/std": 0.12418867647647858, "rewards/judge_quality/mean": 0.3087500035762787, "rewards/judge_quality/std": 0.1141975075006485, "rewards/total_composite/mean": 0.5327460765838623, "rewards/total_composite/std": 0.24773471057415009, "reward": 0.5327460765838623, "reward_std": 0.24773471057415009, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11084035784006119, "sampling/sampling_logp_difference/max": 2.758803367614746, "sampling/importance_sampling_ratio/min": 0.06336755305528641, "sampling/importance_sampling_ratio/mean": 1.0030755996704102, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3492526113986969, "clip_ratio/low_mean": 0.025476908311247826, "clip_ratio/low_min": 0.025476908311247826, "clip_ratio/high_mean": 0.06514233024790883, "clip_ratio/high_max": 0.06514233024790883, "clip_ratio/region_mean": 0.09061923855915666, "reward_total_mean": 0.5327460765838623, "reward_meter_mean": 0.5864174365997314, "reward_meter_std": 0.2772681415081024, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8702847957611084, "reward_repeat_soft_std": 0.12418867647647858, "reward_judge_quality_mean": 0.3087500035762787, "reward_judge_quality_std": 0.1141975075006485, "reward_total_composite_mean": 0.5327460765838623, "reward_total_composite_std": 0.24773471057415009} {"timestamp_utc": "2026-04-13T03:22:24Z", "mode": "train", "global_step": 2013, "epoch": 0.2022099447513812, "loss": -0.0414, "grad_norm": 16.568668365478516, "learning_rate": 3.9030303030303035e-06, "num_tokens": 3688322.0, "completions/mean_length": 23.5, "completions/min_length": 19.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.5, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9098116159439087, "rewards/meter/std": 0.23811741173267365, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9392597675323486, "rewards/repeat_soft/std": 0.04346397891640663, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7815911769866943, "rewards/total_composite/std": 0.10744550079107285, "reward": 0.7815911769866943, "reward_std": 0.10744549334049225, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11678653955459595, "sampling/sampling_logp_difference/max": 1.7698945999145508, "sampling/importance_sampling_ratio/min": 0.17035095393657684, "sampling/importance_sampling_ratio/mean": 1.0088165998458862, "sampling/importance_sampling_ratio/max": 1.7911847829818726, "entropy": 0.6941510066390038, "clip_ratio/low_mean": 0.01315789483487606, "clip_ratio/low_min": 0.01315789483487606, "clip_ratio/high_mean": 0.09672121470794082, "clip_ratio/high_max": 0.09672121470794082, "clip_ratio/region_mean": 0.10987910954281688, "reward_total_mean": 0.7815911769866943, "reward_meter_mean": 0.9098116159439087, "reward_meter_std": 0.23811741173267365, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9392597675323486, "reward_repeat_soft_std": 0.04346397891640663, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7815911769866943, "reward_total_composite_std": 0.10744550079107285} {"timestamp_utc": "2026-04-13T03:22:30Z", "mode": "train", "global_step": 2014, "epoch": 0.20231039678553492, "loss": 0.0573, "grad_norm": 7.811661243438721, "learning_rate": 3.900000000000001e-06, "num_tokens": 3689754.0, "completions/mean_length": 36.0, "completions/min_length": 27.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.0, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.8986053466796875, "rewards/meter/std": 0.2629706859588623, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9379370212554932, "rewards/repeat_soft/std": 0.04684372618794441, "rewards/judge_quality/mean": 0.41749998927116394, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.7734161615371704, "rewards/total_composite/std": 0.11871837079524994, "reward": 0.7734161615371704, "reward_std": 0.11871837079524994, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11799810081720352, "sampling/sampling_logp_difference/max": 2.2516562938690186, "sampling/importance_sampling_ratio/min": 0.26564139127731323, "sampling/importance_sampling_ratio/mean": 1.018944263458252, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6767255663871765, "clip_ratio/low_mean": 0.009146341122686863, "clip_ratio/low_min": 0.009146341122686863, "clip_ratio/high_mean": 0.059998245211318135, "clip_ratio/high_max": 0.059998245211318135, "clip_ratio/region_mean": 0.069144586334005, "reward_total_mean": 0.7734161615371704, "reward_meter_mean": 0.8986053466796875, "reward_meter_std": 0.2629706859588623, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9379370212554932, "reward_repeat_soft_std": 0.04684372618794441, "reward_judge_quality_mean": 0.41749998927116394, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.7734161615371704, "reward_total_composite_std": 0.11871837079524994} {"timestamp_utc": "2026-04-13T03:22:37Z", "mode": "train", "global_step": 2015, "epoch": 0.2024108488196886, "loss": 0.0414, "grad_norm": 10.721487998962402, "learning_rate": 3.896969696969697e-06, "num_tokens": 3691686.0, "completions/mean_length": 60.5, "completions/min_length": 50.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.5, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9024858474731445, "rewards/meter/std": 0.14638064801692963, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9669641852378845, "rewards/repeat_soft/std": 0.02762051485478878, "rewards/judge_quality/mean": 0.32499998807907104, "rewards/judge_quality/std": 0.1035098284482956, "rewards/total_composite/mean": 0.7503150701522827, "rewards/total_composite/std": 0.05001826956868172, "reward": 0.7503150701522827, "reward_std": 0.05001826584339142, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.102566197514534, "sampling/sampling_logp_difference/max": 1.9087491035461426, "sampling/importance_sampling_ratio/min": 0.14826573431491852, "sampling/importance_sampling_ratio/mean": 1.0079853534698486, "sampling/importance_sampling_ratio/max": 1.997775912284851, "entropy": 0.6251544840633869, "clip_ratio/low_mean": 0.019513826817274094, "clip_ratio/low_min": 0.019513826817274094, "clip_ratio/high_mean": 0.0742036565206945, "clip_ratio/high_max": 0.0742036565206945, "clip_ratio/region_mean": 0.09371748333796859, "reward_total_mean": 0.7503150701522827, "reward_meter_mean": 0.9024858474731445, "reward_meter_std": 0.14638064801692963, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9669641852378845, "reward_repeat_soft_std": 0.02762051485478878, "reward_judge_quality_mean": 0.32499998807907104, "reward_judge_quality_std": 0.1035098284482956, "reward_total_composite_mean": 0.7503150701522827, "reward_total_composite_std": 0.05001826956868172} {"timestamp_utc": "2026-04-13T03:22:43Z", "mode": "train", "global_step": 2016, "epoch": 0.20251130085384228, "loss": -0.0099, "grad_norm": 9.132987976074219, "learning_rate": 3.8939393939393944e-06, "num_tokens": 3693494.0, "completions/mean_length": 57.0, "completions/min_length": 51.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.0, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9033676385879517, "rewards/meter/std": 0.22317011654376984, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9256414175033569, "rewards/repeat_soft/std": 0.0387575738132, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.7698295712471008, "rewards/total_composite/std": 0.10077675431966782, "reward": 0.7698295712471008, "reward_std": 0.10077673941850662, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09866682440042496, "sampling/sampling_logp_difference/max": 1.257166862487793, "sampling/importance_sampling_ratio/min": 0.28445881605148315, "sampling/importance_sampling_ratio/mean": 1.0127431154251099, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.615705780684948, "clip_ratio/low_mean": 0.014517345931380987, "clip_ratio/low_min": 0.014517345931380987, "clip_ratio/high_mean": 0.050898175686597824, "clip_ratio/high_max": 0.050898175686597824, "clip_ratio/region_mean": 0.06541552161797881, "reward_total_mean": 0.7698295712471008, "reward_meter_mean": 0.9033676385879517, "reward_meter_std": 0.22317011654376984, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9256414175033569, "reward_repeat_soft_std": 0.0387575738132, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.7698295712471008, "reward_total_composite_std": 0.10077675431966782} {"timestamp_utc": "2026-04-13T03:22:54Z", "mode": "train", "global_step": 2017, "epoch": 0.202611752887996, "loss": 0.0032, "grad_norm": 5.902063846588135, "learning_rate": 3.890909090909092e-06, "num_tokens": 3696119.0, "completions/mean_length": 149.125, "completions/min_length": 133.0, "completions/max_length": 165.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 149.125, "completions/min_terminated_length": 133.0, "completions/max_terminated_length": 165.0, "rewards/meter/mean": 0.9923721551895142, "rewards/meter/std": 0.003516758093610406, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8955735564231873, "rewards/repeat_soft/std": 0.04717658460140228, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7883123159408569, "rewards/total_composite/std": 0.033334311097860336, "reward": 0.7883123159408569, "reward_std": 0.033334340900182724, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10367952287197113, "sampling/sampling_logp_difference/max": 1.3753066062927246, "sampling/importance_sampling_ratio/min": 0.2527620792388916, "sampling/importance_sampling_ratio/mean": 1.0065631866455078, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6216148398816586, "clip_ratio/low_mean": 0.0393364280462265, "clip_ratio/low_min": 0.0393364280462265, "clip_ratio/high_mean": 0.05672670993953943, "clip_ratio/high_max": 0.05672670993953943, "clip_ratio/region_mean": 0.09606313798576593, "reward_total_mean": 0.7883123159408569, "reward_meter_mean": 0.9923721551895142, "reward_meter_std": 0.003516758093610406, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8955735564231873, "reward_repeat_soft_std": 0.04717658460140228, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7883123159408569, "reward_total_composite_std": 0.033334311097860336} {"timestamp_utc": "2026-04-13T03:23:01Z", "mode": "train", "global_step": 2018, "epoch": 0.20271220492214967, "loss": -0.0967, "grad_norm": 15.92779541015625, "learning_rate": 3.887878787878788e-06, "num_tokens": 3697749.0, "completions/mean_length": 39.75, "completions/min_length": 28.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.75, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.8303543329238892, "rewards/meter/std": 0.17569181323051453, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9779375791549683, "rewards/repeat_soft/std": 0.01813257299363613, "rewards/judge_quality/mean": 0.7362500429153442, "rewards/judge_quality/std": 0.25376805663108826, "rewards/total_composite/mean": 0.8423281908035278, "rewards/total_composite/std": 0.12339355796575546, "reward": 0.8423281908035278, "reward_std": 0.12339355051517487, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12640579044818878, "sampling/sampling_logp_difference/max": 1.0161633491516113, "sampling/importance_sampling_ratio/min": 0.3619810938835144, "sampling/importance_sampling_ratio/mean": 1.0091235637664795, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7016246020793915, "clip_ratio/low_mean": 0.08308913465589285, "clip_ratio/low_min": 0.08308913465589285, "clip_ratio/high_mean": 0.08823665417730808, "clip_ratio/high_max": 0.08823665417730808, "clip_ratio/region_mean": 0.17132578883320093, "reward_total_mean": 0.8423281908035278, "reward_meter_mean": 0.8303543329238892, "reward_meter_std": 0.17569181323051453, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9779375791549683, "reward_repeat_soft_std": 0.01813257299363613, "reward_judge_quality_mean": 0.7362500429153442, "reward_judge_quality_std": 0.25376805663108826, "reward_total_composite_mean": 0.8423281908035278, "reward_total_composite_std": 0.12339355796575546} {"timestamp_utc": "2026-04-13T03:23:08Z", "mode": "train", "global_step": 2019, "epoch": 0.20281265695630338, "loss": -0.0155, "grad_norm": 18.49652862548828, "learning_rate": 3.884848484848485e-06, "num_tokens": 3699603.0, "completions/mean_length": 54.75, "completions/min_length": 49.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.75, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.897741436958313, "rewards/meter/std": 0.2517523765563965, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9363080859184265, "rewards/repeat_soft/std": 0.046849824488162994, "rewards/judge_quality/mean": 0.3174999952316284, "rewards/judge_quality/std": 0.0936177596449852, "rewards/total_composite/mean": 0.7428644895553589, "rewards/total_composite/std": 0.10299307852983475, "reward": 0.7428644895553589, "reward_std": 0.10299309343099594, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11152128875255585, "sampling/sampling_logp_difference/max": 1.5637092590332031, "sampling/importance_sampling_ratio/min": 0.20935805141925812, "sampling/importance_sampling_ratio/mean": 1.0108476877212524, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6814202032983303, "clip_ratio/low_mean": 0.025510204955935478, "clip_ratio/low_min": 0.025510204955935478, "clip_ratio/high_mean": 0.08566789934411645, "clip_ratio/high_max": 0.08566789934411645, "clip_ratio/region_mean": 0.11117810430005193, "reward_total_mean": 0.7428644895553589, "reward_meter_mean": 0.897741436958313, "reward_meter_std": 0.2517523765563965, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9363080859184265, "reward_repeat_soft_std": 0.046849824488162994, "reward_judge_quality_mean": 0.3174999952316284, "reward_judge_quality_std": 0.0936177596449852, "reward_total_composite_mean": 0.7428644895553589, "reward_total_composite_std": 0.10299307852983475} {"timestamp_utc": "2026-04-13T03:23:16Z", "mode": "train", "global_step": 2020, "epoch": 0.20291310899045706, "loss": 0.0086, "grad_norm": 12.576126098632812, "learning_rate": 3.881818181818182e-06, "num_tokens": 3701874.0, "completions/mean_length": 101.875, "completions/min_length": 93.0, "completions/max_length": 110.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.875, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 110.0, "rewards/meter/mean": 0.9719498753547668, "rewards/meter/std": 0.039483606815338135, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9170314073562622, "rewards/repeat_soft/std": 0.06050790101289749, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.6910706758499146, "rewards/total_composite/std": 0.28059402108192444, "reward": 0.6910706758499146, "reward_std": 0.28059402108192444, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09019721299409866, "sampling/sampling_logp_difference/max": 1.2972168922424316, "sampling/importance_sampling_ratio/min": 0.27329131960868835, "sampling/importance_sampling_ratio/mean": 0.9957608580589294, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5578687116503716, "clip_ratio/low_mean": 0.009210526011884212, "clip_ratio/low_min": 0.009210526011884212, "clip_ratio/high_mean": 0.08785907691344619, "clip_ratio/high_max": 0.08785907691344619, "clip_ratio/region_mean": 0.0970696029253304, "reward_total_mean": 0.6910706758499146, "reward_meter_mean": 0.9719498753547668, "reward_meter_std": 0.039483606815338135, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9170314073562622, "reward_repeat_soft_std": 0.06050790101289749, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.6910706758499146, "reward_total_composite_std": 0.28059402108192444} {"timestamp_utc": "2026-04-13T03:23:22Z", "mode": "train", "global_step": 2021, "epoch": 0.20301356102461074, "loss": 0.0342, "grad_norm": 11.190730094909668, "learning_rate": 3.878787878787879e-06, "num_tokens": 3703462.0, "completions/mean_length": 56.5, "completions/min_length": 50.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.5, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.7387773990631104, "rewards/meter/std": 0.34287548065185547, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9850406646728516, "rewards/repeat_soft/std": 0.014333167113363743, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.7455788850784302, "rewards/total_composite/std": 0.16078054904937744, "reward": 0.7455788850784302, "reward_std": 0.16078054904937744, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11825212091207504, "sampling/sampling_logp_difference/max": 1.3198938369750977, "sampling/importance_sampling_ratio/min": 0.2671636641025543, "sampling/importance_sampling_ratio/mean": 1.0074245929718018, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6424197554588318, "clip_ratio/low_mean": 0.051274086348712444, "clip_ratio/low_min": 0.051274086348712444, "clip_ratio/high_mean": 0.07887696195393801, "clip_ratio/high_max": 0.07887696195393801, "clip_ratio/region_mean": 0.13015104830265045, "reward_total_mean": 0.7455788850784302, "reward_meter_mean": 0.7387773990631104, "reward_meter_std": 0.34287548065185547, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9850406646728516, "reward_repeat_soft_std": 0.014333167113363743, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.7455788850784302, "reward_total_composite_std": 0.16078054904937744} {"timestamp_utc": "2026-04-13T03:23:30Z", "mode": "train", "global_step": 2022, "epoch": 0.20311401305876445, "loss": 0.0987, "grad_norm": 7.351996898651123, "learning_rate": 3.875757575757576e-06, "num_tokens": 3705950.0, "completions/mean_length": 135.0, "completions/min_length": 112.0, "completions/max_length": 155.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 135.0, "completions/min_terminated_length": 112.0, "completions/max_terminated_length": 155.0, "rewards/meter/mean": 0.9582208395004272, "rewards/meter/std": 0.01784423552453518, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8888227939605713, "rewards/repeat_soft/std": 0.05593082681298256, "rewards/judge_quality/mean": 0.17500001192092896, "rewards/judge_quality/std": 0.04629100486636162, "rewards/total_composite/mean": 0.7132066488265991, "rewards/total_composite/std": 0.020158838480710983, "reward": 0.7132066488265991, "reward_std": 0.02015884406864643, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0944233313202858, "sampling/sampling_logp_difference/max": 1.8010401725769043, "sampling/importance_sampling_ratio/min": 0.16512705385684967, "sampling/importance_sampling_ratio/mean": 1.013421893119812, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5374343432486057, "clip_ratio/low_mean": 0.030993151012808084, "clip_ratio/low_min": 0.030993151012808084, "clip_ratio/high_mean": 0.04809366539120674, "clip_ratio/high_max": 0.04809366539120674, "clip_ratio/region_mean": 0.07908681640401483, "reward_total_mean": 0.7132066488265991, "reward_meter_mean": 0.9582208395004272, "reward_meter_std": 0.01784423552453518, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8888227939605713, "reward_repeat_soft_std": 0.05593082681298256, "reward_judge_quality_mean": 0.17500001192092896, "reward_judge_quality_std": 0.04629100486636162, "reward_total_composite_mean": 0.7132066488265991, "reward_total_composite_std": 0.020158838480710983} {"timestamp_utc": "2026-04-13T03:23:36Z", "mode": "train", "global_step": 2023, "epoch": 0.20321446509291813, "loss": 0.0005, "grad_norm": 11.097015380859375, "learning_rate": 3.872727272727273e-06, "num_tokens": 3707779.0, "completions/mean_length": 63.625, "completions/min_length": 56.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.625, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9838067293167114, "rewards/meter/std": 0.012381844222545624, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9036665558815002, "rewards/repeat_soft/std": 0.0563390888273716, "rewards/judge_quality/mean": 0.3687499761581421, "rewards/judge_quality/std": 0.10802611708641052, "rewards/total_composite/mean": 0.7937046885490417, "rewards/total_composite/std": 0.038001492619514465, "reward": 0.7937046885490417, "reward_std": 0.03800148889422417, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09654522687196732, "sampling/sampling_logp_difference/max": 3.180936336517334, "sampling/importance_sampling_ratio/min": 0.04154673591256142, "sampling/importance_sampling_ratio/mean": 1.018983244895935, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5098957568407059, "clip_ratio/low_mean": 0.02784090954810381, "clip_ratio/low_min": 0.02784090954810381, "clip_ratio/high_mean": 0.06893174396827817, "clip_ratio/high_max": 0.06893174396827817, "clip_ratio/region_mean": 0.09677265351638198, "reward_total_mean": 0.7937046885490417, "reward_meter_mean": 0.9838067293167114, "reward_meter_std": 0.012381844222545624, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9036665558815002, "reward_repeat_soft_std": 0.0563390888273716, "reward_judge_quality_mean": 0.3687499761581421, "reward_judge_quality_std": 0.10802611708641052, "reward_total_composite_mean": 0.7937046885490417, "reward_total_composite_std": 0.038001492619514465} {"timestamp_utc": "2026-04-13T03:23:48Z", "mode": "train", "global_step": 2024, "epoch": 0.20331491712707184, "loss": -0.1135, "grad_norm": 1.7564641237258911, "learning_rate": 3.86969696969697e-06, "num_tokens": 3709388.0, "completions/mean_length": 102.125, "completions/min_length": 38.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 43.57143020629883, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.851691722869873, "rewards/meter/std": 0.16028831899166107, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9693377017974854, "rewards/repeat_soft/std": 0.03840930014848709, "rewards/judge_quality/mean": 0.38875001668930054, "rewards/judge_quality/std": 0.13767844438552856, "rewards/total_composite/mean": 0.6603077054023743, "rewards/total_composite/std": 0.2767300307750702, "reward": 0.6603077054023743, "reward_std": 0.2767300009727478, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1137702465057373, "sampling/sampling_logp_difference/max": 1.2996996641159058, "sampling/importance_sampling_ratio/min": 0.27261364459991455, "sampling/importance_sampling_ratio/mean": 0.9909107685089111, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4840565323829651, "clip_ratio/low_mean": 0.027795161120593548, "clip_ratio/low_min": 0.027795161120593548, "clip_ratio/high_mean": 0.0992510924115777, "clip_ratio/high_max": 0.0992510924115777, "clip_ratio/region_mean": 0.12704625353217125, "reward_total_mean": 0.6603077054023743, "reward_meter_mean": 0.851691722869873, "reward_meter_std": 0.16028831899166107, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9693377017974854, "reward_repeat_soft_std": 0.03840930014848709, "reward_judge_quality_mean": 0.38875001668930054, "reward_judge_quality_std": 0.13767844438552856, "reward_total_composite_mean": 0.6603077054023743, "reward_total_composite_std": 0.2767300307750702} {"timestamp_utc": "2026-04-13T03:23:55Z", "mode": "train", "global_step": 2025, "epoch": 0.20341536916122552, "loss": -0.0151, "grad_norm": 8.354020118713379, "learning_rate": 3.866666666666667e-06, "num_tokens": 3711211.0, "completions/mean_length": 60.875, "completions/min_length": 57.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9739528894424438, "rewards/meter/std": 0.0506206676363945, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9766963720321655, "rewards/repeat_soft/std": 0.019637862220406532, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.8066984415054321, "rewards/total_composite/std": 0.029455069452524185, "reward": 0.8066984415054321, "reward_std": 0.029455067589879036, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09095009416341782, "sampling/sampling_logp_difference/max": 1.4443349838256836, "sampling/importance_sampling_ratio/min": 0.23590292036533356, "sampling/importance_sampling_ratio/mean": 1.000968098640442, "sampling/importance_sampling_ratio/max": 1.8701854944229126, "entropy": 0.529117152094841, "clip_ratio/low_mean": 0.01950998231768608, "clip_ratio/low_min": 0.01950998231768608, "clip_ratio/high_mean": 0.06478135101497173, "clip_ratio/high_max": 0.06478135101497173, "clip_ratio/region_mean": 0.08429133333265781, "reward_total_mean": 0.8066984415054321, "reward_meter_mean": 0.9739528894424438, "reward_meter_std": 0.0506206676363945, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9766963720321655, "reward_repeat_soft_std": 0.019637862220406532, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.8066984415054321, "reward_total_composite_std": 0.029455069452524185} {"timestamp_utc": "2026-04-13T03:24:01Z", "mode": "train", "global_step": 2026, "epoch": 0.2035158211953792, "loss": 0.0773, "grad_norm": 10.443721771240234, "learning_rate": 3.863636363636364e-06, "num_tokens": 3712854.0, "completions/mean_length": 58.375, "completions/min_length": 45.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.375, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.530418336391449, "rewards/meter/std": 0.4332382082939148, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9640227556228638, "rewards/repeat_soft/std": 0.041732512414455414, "rewards/judge_quality/mean": 0.44749999046325684, "rewards/judge_quality/std": 0.20824094116687775, "rewards/total_composite/mean": 0.6193405389785767, "rewards/total_composite/std": 0.17250801622867584, "reward": 0.6193405389785767, "reward_std": 0.17250801622867584, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13232539594173431, "sampling/sampling_logp_difference/max": 2.396242141723633, "sampling/importance_sampling_ratio/min": 0.09105949848890305, "sampling/importance_sampling_ratio/mean": 1.0018516778945923, "sampling/importance_sampling_ratio/max": 1.8924922943115234, "entropy": 0.9089943841099739, "clip_ratio/low_mean": 0.04868421144783497, "clip_ratio/low_min": 0.04868421144783497, "clip_ratio/high_mean": 0.08549066446721554, "clip_ratio/high_max": 0.08549066446721554, "clip_ratio/region_mean": 0.1341748759150505, "reward_total_mean": 0.6193405389785767, "reward_meter_mean": 0.530418336391449, "reward_meter_std": 0.4332382082939148, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9640227556228638, "reward_repeat_soft_std": 0.041732512414455414, "reward_judge_quality_mean": 0.44749999046325684, "reward_judge_quality_std": 0.20824094116687775, "reward_total_composite_mean": 0.6193405389785767, "reward_total_composite_std": 0.17250801622867584} {"timestamp_utc": "2026-04-13T03:24:08Z", "mode": "train", "global_step": 2027, "epoch": 0.2036162732295329, "loss": 0.1352, "grad_norm": 7.761141300201416, "learning_rate": 3.860606060606061e-06, "num_tokens": 3714505.0, "completions/mean_length": 53.375, "completions/min_length": 44.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.375, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9766459465026855, "rewards/meter/std": 0.00965864583849907, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8767869472503662, "rewards/repeat_soft/std": 0.07275299727916718, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.8080443143844604, "rewards/total_composite/std": 0.04380928725004196, "reward": 0.8080443143844604, "reward_std": 0.04380929097533226, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08071890473365784, "sampling/sampling_logp_difference/max": 1.7792658805847168, "sampling/importance_sampling_ratio/min": 0.1687619984149933, "sampling/importance_sampling_ratio/mean": 0.9930741190910339, "sampling/importance_sampling_ratio/max": 1.5800440311431885, "entropy": 0.39492112025618553, "clip_ratio/low_mean": 0.0353806740604341, "clip_ratio/low_min": 0.0353806740604341, "clip_ratio/high_mean": 0.0493897320702672, "clip_ratio/high_max": 0.0493897320702672, "clip_ratio/region_mean": 0.0847704061307013, "reward_total_mean": 0.8080443143844604, "reward_meter_mean": 0.9766459465026855, "reward_meter_std": 0.00965864583849907, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8767869472503662, "reward_repeat_soft_std": 0.07275299727916718, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.8080443143844604, "reward_total_composite_std": 0.04380928725004196} {"timestamp_utc": "2026-04-13T03:24:17Z", "mode": "train", "global_step": 2028, "epoch": 0.2037167252636866, "loss": 0.0024, "grad_norm": 6.571693420410156, "learning_rate": 3.857575757575758e-06, "num_tokens": 3717526.0, "completions/mean_length": 186.625, "completions/min_length": 173.0, "completions/max_length": 210.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 186.625, "completions/min_terminated_length": 173.0, "completions/max_terminated_length": 210.0, "rewards/meter/mean": 0.9848736524581909, "rewards/meter/std": 0.011868826113641262, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9159432649612427, "rewards/repeat_soft/std": 0.02873641438782215, "rewards/judge_quality/mean": 0.23749999701976776, "rewards/judge_quality/std": 0.0353553369641304, "rewards/total_composite/mean": 0.7560374736785889, "rewards/total_composite/std": 0.012193984352052212, "reward": 0.7560374736785889, "reward_std": 0.012193959206342697, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10833075642585754, "sampling/sampling_logp_difference/max": 1.8295230865478516, "sampling/importance_sampling_ratio/min": 0.16049009561538696, "sampling/importance_sampling_ratio/mean": 1.01465904712677, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6845813244581223, "clip_ratio/low_mean": 0.03137046378105879, "clip_ratio/low_min": 0.03137046378105879, "clip_ratio/high_mean": 0.06133668310940266, "clip_ratio/high_max": 0.06133668310940266, "clip_ratio/region_mean": 0.09270714689046144, "reward_total_mean": 0.7560374736785889, "reward_meter_mean": 0.9848736524581909, "reward_meter_std": 0.011868826113641262, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9159432649612427, "reward_repeat_soft_std": 0.02873641438782215, "reward_judge_quality_mean": 0.23749999701976776, "reward_judge_quality_std": 0.0353553369641304, "reward_total_composite_mean": 0.7560374736785889, "reward_total_composite_std": 0.012193984352052212} {"timestamp_utc": "2026-04-13T03:24:24Z", "mode": "train", "global_step": 2029, "epoch": 0.20381717729784027, "loss": 0.0238, "grad_norm": 6.44549036026001, "learning_rate": 3.8545454545454545e-06, "num_tokens": 3719895.0, "completions/mean_length": 138.125, "completions/min_length": 120.0, "completions/max_length": 156.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 138.125, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 156.0, "rewards/meter/mean": 0.859426736831665, "rewards/meter/std": 0.1444167196750641, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9111765623092651, "rewards/repeat_soft/std": 0.05096001923084259, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.12351980805397034, "rewards/total_composite/mean": 0.7171096801757812, "rewards/total_composite/std": 0.051079485565423965, "reward": 0.7171096801757812, "reward_std": 0.051079485565423965, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09654983133077621, "sampling/sampling_logp_difference/max": 1.8473691940307617, "sampling/importance_sampling_ratio/min": 0.1576513648033142, "sampling/importance_sampling_ratio/mean": 1.0067689418792725, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.46965524181723595, "clip_ratio/low_mean": 0.04573863744735718, "clip_ratio/low_min": 0.04573863744735718, "clip_ratio/high_mean": 0.03728489577770233, "clip_ratio/high_max": 0.03728489577770233, "clip_ratio/region_mean": 0.08302353322505951, "reward_total_mean": 0.7171096801757812, "reward_meter_mean": 0.859426736831665, "reward_meter_std": 0.1444167196750641, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9111765623092651, "reward_repeat_soft_std": 0.05096001923084259, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.12351980805397034, "reward_total_composite_mean": 0.7171096801757812, "reward_total_composite_std": 0.051079485565423965} {"timestamp_utc": "2026-04-13T03:24:31Z", "mode": "train", "global_step": 2030, "epoch": 0.20391762933199398, "loss": 0.0641, "grad_norm": 7.728828430175781, "learning_rate": 3.851515151515152e-06, "num_tokens": 3721602.0, "completions/mean_length": 60.375, "completions/min_length": 52.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.375, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9713665246963501, "rewards/meter/std": 0.01983739621937275, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9654348492622375, "rewards/repeat_soft/std": 0.029399529099464417, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8107834458351135, "rewards/total_composite/std": 0.006258830893784761, "reward": 0.8107834458351135, "reward_std": 0.006258820183575153, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09449785947799683, "sampling/sampling_logp_difference/max": 1.3671579360961914, "sampling/importance_sampling_ratio/min": 0.25483018159866333, "sampling/importance_sampling_ratio/mean": 1.0009466409683228, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5423296429216862, "clip_ratio/low_mean": 0.04943973198533058, "clip_ratio/low_min": 0.04943973198533058, "clip_ratio/high_mean": 0.05209688376635313, "clip_ratio/high_max": 0.05209688376635313, "clip_ratio/region_mean": 0.10153661575168371, "reward_total_mean": 0.8107834458351135, "reward_meter_mean": 0.9713665246963501, "reward_meter_std": 0.01983739621937275, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9654348492622375, "reward_repeat_soft_std": 0.029399529099464417, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8107834458351135, "reward_total_composite_std": 0.006258830893784761} {"timestamp_utc": "2026-04-13T03:24:38Z", "mode": "train", "global_step": 2031, "epoch": 0.20401808136614766, "loss": 0.0436, "grad_norm": 12.656648635864258, "learning_rate": 3.848484848484848e-06, "num_tokens": 3723150.0, "completions/mean_length": 30.5, "completions/min_length": 28.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.5, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9496431350708008, "rewards/meter/std": 0.034699130803346634, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9562410116195679, "rewards/repeat_soft/std": 0.011590460315346718, "rewards/judge_quality/mean": 0.41749998927116394, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.798213541507721, "rewards/total_composite/std": 0.026786914095282555, "reward": 0.798213541507721, "reward_std": 0.026786917820572853, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12476413697004318, "sampling/sampling_logp_difference/max": 1.222041130065918, "sampling/importance_sampling_ratio/min": 0.29462817311286926, "sampling/importance_sampling_ratio/mean": 0.9881458282470703, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5992751121520996, "clip_ratio/low_mean": 0.04767839703708887, "clip_ratio/low_min": 0.04767839703708887, "clip_ratio/high_mean": 0.0809381459839642, "clip_ratio/high_max": 0.0809381459839642, "clip_ratio/region_mean": 0.12861654302105308, "reward_total_mean": 0.798213541507721, "reward_meter_mean": 0.9496431350708008, "reward_meter_std": 0.034699130803346634, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9562410116195679, "reward_repeat_soft_std": 0.011590460315346718, "reward_judge_quality_mean": 0.41749998927116394, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.798213541507721, "reward_total_composite_std": 0.026786914095282555} {"timestamp_utc": "2026-04-13T03:24:44Z", "mode": "train", "global_step": 2032, "epoch": 0.20411853340030137, "loss": 0.0545, "grad_norm": 14.867531776428223, "learning_rate": 3.8454545454545454e-06, "num_tokens": 3724679.0, "completions/mean_length": 30.125, "completions/min_length": 27.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.125, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.994915783405304, "rewards/meter/std": 0.002773250686004758, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7782723903656006, "rewards/repeat_soft/std": 0.10849412530660629, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.7974143028259277, "rewards/total_composite/std": 0.015910807996988297, "reward": 0.7974143028259277, "reward_std": 0.01591080240905285, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08228503167629242, "sampling/sampling_logp_difference/max": 0.9828906059265137, "sampling/importance_sampling_ratio/min": 0.3742277920246124, "sampling/importance_sampling_ratio/mean": 1.0120019912719727, "sampling/importance_sampling_ratio/max": 1.6953152418136597, "entropy": 0.43470459803938866, "clip_ratio/low_mean": 0.03533724416047335, "clip_ratio/low_min": 0.03533724416047335, "clip_ratio/high_mean": 0.042187233455479145, "clip_ratio/high_max": 0.042187233455479145, "clip_ratio/region_mean": 0.07752447761595249, "reward_total_mean": 0.7974143028259277, "reward_meter_mean": 0.994915783405304, "reward_meter_std": 0.002773250686004758, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7782723903656006, "reward_repeat_soft_std": 0.10849412530660629, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.7974143028259277, "reward_total_composite_std": 0.015910807996988297} {"timestamp_utc": "2026-04-13T03:24:56Z", "mode": "train", "global_step": 2033, "epoch": 0.20421898543445505, "loss": -0.236, "grad_norm": 1.2896226644515991, "learning_rate": 3.842424242424243e-06, "num_tokens": 3727184.0, "completions/mean_length": 203.125, "completions/min_length": 145.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 159.0, "completions/min_terminated_length": 145.0, "completions/max_terminated_length": 179.0, "rewards/meter/mean": 0.8399412631988525, "rewards/meter/std": 0.33673012256622314, "rewards/count_adherence/mean": 0.925000011920929, "rewards/count_adherence/std": 0.2121320217847824, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8460023999214172, "rewards/repeat_soft/std": 0.07057762145996094, "rewards/judge_quality/mean": 0.17500001192092896, "rewards/judge_quality/std": 0.0707106813788414, "rewards/total_composite/mean": 0.6323497295379639, "rewards/total_composite/std": 0.25637850165367126, "reward": 0.6323497295379639, "reward_std": 0.25637850165367126, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07443364709615707, "sampling/sampling_logp_difference/max": 1.9823321104049683, "sampling/importance_sampling_ratio/min": 0.1377476155757904, "sampling/importance_sampling_ratio/mean": 1.005185842514038, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3834143988788128, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0668712523765862, "clip_ratio/high_max": 0.0668712523765862, "clip_ratio/region_mean": 0.0668712523765862, "reward_total_mean": 0.6323497295379639, "reward_meter_mean": 0.8399412631988525, "reward_meter_std": 0.33673012256622314, "reward_count_adherence_mean": 0.925000011920929, "reward_count_adherence_std": 0.2121320217847824, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8460023999214172, "reward_repeat_soft_std": 0.07057762145996094, "reward_judge_quality_mean": 0.17500001192092896, "reward_judge_quality_std": 0.0707106813788414, "reward_total_composite_mean": 0.6323497295379639, "reward_total_composite_std": 0.25637850165367126} {"timestamp_utc": "2026-04-13T03:25:03Z", "mode": "train", "global_step": 2034, "epoch": 0.20431943746860873, "loss": 0.04, "grad_norm": 9.506256103515625, "learning_rate": 3.839393939393939e-06, "num_tokens": 3729129.0, "completions/mean_length": 73.125, "completions/min_length": 61.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.125, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.9750977158546448, "rewards/meter/std": 0.02229495532810688, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8693511486053467, "rewards/repeat_soft/std": 0.08544005453586578, "rewards/judge_quality/mean": 0.2799999713897705, "rewards/judge_quality/std": 0.09304375946521759, "rewards/total_composite/mean": 0.7597290277481079, "rewards/total_composite/std": 0.028656329959630966, "reward": 0.7597290277481079, "reward_std": 0.028656329959630966, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08463344722986221, "sampling/sampling_logp_difference/max": 1.252028465270996, "sampling/importance_sampling_ratio/min": 0.40562742948532104, "sampling/importance_sampling_ratio/mean": 1.0198200941085815, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5872636809945107, "clip_ratio/low_mean": 0.054390822537243366, "clip_ratio/low_min": 0.054390822537243366, "clip_ratio/high_mean": 0.020051533356308937, "clip_ratio/high_max": 0.020051533356308937, "clip_ratio/region_mean": 0.0744423558935523, "reward_total_mean": 0.7597290277481079, "reward_meter_mean": 0.9750977158546448, "reward_meter_std": 0.02229495532810688, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8693511486053467, "reward_repeat_soft_std": 0.08544005453586578, "reward_judge_quality_mean": 0.2799999713897705, "reward_judge_quality_std": 0.09304375946521759, "reward_total_composite_mean": 0.7597290277481079, "reward_total_composite_std": 0.028656329959630966} {"timestamp_utc": "2026-04-13T03:25:10Z", "mode": "train", "global_step": 2035, "epoch": 0.20441988950276244, "loss": 0.0032, "grad_norm": 6.134500503540039, "learning_rate": 3.836363636363636e-06, "num_tokens": 3731539.0, "completions/mean_length": 114.25, "completions/min_length": 106.0, "completions/max_length": 119.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.25, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.9807287454605103, "rewards/meter/std": 0.007146927528083324, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.925352931022644, "rewards/repeat_soft/std": 0.047040484845638275, "rewards/judge_quality/mean": 0.29249998927116394, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7716132402420044, "rewards/total_composite/std": 0.025135602802038193, "reward": 0.7716132402420044, "reward_std": 0.025135576725006104, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08730481564998627, "sampling/sampling_logp_difference/max": 2.087258815765381, "sampling/importance_sampling_ratio/min": 0.12402664124965668, "sampling/importance_sampling_ratio/mean": 1.004267930984497, "sampling/importance_sampling_ratio/max": 1.903458833694458, "entropy": 0.5229752026498318, "clip_ratio/low_mean": 0.05679879104718566, "clip_ratio/low_min": 0.05679879104718566, "clip_ratio/high_mean": 0.02060297830030322, "clip_ratio/high_max": 0.02060297830030322, "clip_ratio/region_mean": 0.07740176934748888, "reward_total_mean": 0.7716132402420044, "reward_meter_mean": 0.9807287454605103, "reward_meter_std": 0.007146927528083324, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.925352931022644, "reward_repeat_soft_std": 0.047040484845638275, "reward_judge_quality_mean": 0.29249998927116394, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7716132402420044, "reward_total_composite_std": 0.025135602802038193} {"timestamp_utc": "2026-04-13T03:25:22Z", "mode": "train", "global_step": 2036, "epoch": 0.20452034153691612, "loss": -0.2092, "grad_norm": 1.8864123821258545, "learning_rate": 3.833333333333334e-06, "num_tokens": 3734091.0, "completions/mean_length": 195.0, "completions/min_length": 141.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 149.71429443359375, "completions/min_terminated_length": 141.0, "completions/max_terminated_length": 172.0, "rewards/meter/mean": 0.83001309633255, "rewards/meter/std": 0.2581445872783661, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8989782929420471, "rewards/repeat_soft/std": 0.09567328542470932, "rewards/judge_quality/mean": 0.26749998331069946, "rewards/judge_quality/std": 0.11671087145805359, "rewards/total_composite/mean": 0.6056826114654541, "rewards/total_composite/std": 0.25980687141418457, "reward": 0.6056826114654541, "reward_std": 0.25980687141418457, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08864624053239822, "sampling/sampling_logp_difference/max": 2.063096761703491, "sampling/importance_sampling_ratio/min": 0.12705987691879272, "sampling/importance_sampling_ratio/mean": 1.0091850757598877, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48316601291298866, "clip_ratio/low_mean": 0.023597237654030323, "clip_ratio/low_min": 0.023597237654030323, "clip_ratio/high_mean": 0.04640236729755998, "clip_ratio/high_max": 0.04640236729755998, "clip_ratio/region_mean": 0.0699996049515903, "reward_total_mean": 0.6056826114654541, "reward_meter_mean": 0.83001309633255, "reward_meter_std": 0.2581445872783661, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.2651650309562683, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8989782929420471, "reward_repeat_soft_std": 0.09567328542470932, "reward_judge_quality_mean": 0.26749998331069946, "reward_judge_quality_std": 0.11671087145805359, "reward_total_composite_mean": 0.6056826114654541, "reward_total_composite_std": 0.25980687141418457} {"timestamp_utc": "2026-04-13T03:25:34Z", "mode": "train", "global_step": 2037, "epoch": 0.20462079357106983, "loss": -0.1199, "grad_norm": 2.4864892959594727, "learning_rate": 3.830303030303031e-06, "num_tokens": 3735818.0, "completions/mean_length": 113.875, "completions/min_length": 49.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 57.000003814697266, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.4825296401977539, "rewards/meter/std": 0.32121628522872925, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9907963275909424, "rewards/repeat_soft/std": 0.012967847287654877, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.5450239181518555, "rewards/total_composite/std": 0.24908195436000824, "reward": 0.5450239181518555, "reward_std": 0.24908195436000824, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11192212998867035, "sampling/sampling_logp_difference/max": 4.4509782791137695, "sampling/importance_sampling_ratio/min": 0.01166714821010828, "sampling/importance_sampling_ratio/mean": 0.9928629994392395, "sampling/importance_sampling_ratio/max": 1.5176187753677368, "entropy": 0.5302727818489075, "clip_ratio/low_mean": 0.010458703385666013, "clip_ratio/low_min": 0.010458703385666013, "clip_ratio/high_mean": 0.09632007870823145, "clip_ratio/high_max": 0.09632007870823145, "clip_ratio/region_mean": 0.10677878209389746, "reward_total_mean": 0.5450239181518555, "reward_meter_mean": 0.4825296401977539, "reward_meter_std": 0.32121628522872925, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9907963275909424, "reward_repeat_soft_std": 0.012967847287654877, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.5450239181518555, "reward_total_composite_std": 0.24908195436000824} {"timestamp_utc": "2026-04-13T03:25:40Z", "mode": "train", "global_step": 2038, "epoch": 0.2047212456052235, "loss": -0.0327, "grad_norm": 17.545513153076172, "learning_rate": 3.827272727272728e-06, "num_tokens": 3737372.0, "completions/mean_length": 34.25, "completions/min_length": 30.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.25, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.27078092098236084, "rewards/meter/std": 0.31442588567733765, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9928053617477417, "rewards/repeat_soft/std": 0.012530048377811909, "rewards/judge_quality/mean": 0.48374998569488525, "rewards/judge_quality/std": 0.18965664505958557, "rewards/total_composite/mean": 0.45879900455474854, "rewards/total_composite/std": 0.2301139533519745, "reward": 0.45879900455474854, "reward_std": 0.2301139533519745, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1096913143992424, "sampling/sampling_logp_difference/max": 1.7795367240905762, "sampling/importance_sampling_ratio/min": 0.16871629655361176, "sampling/importance_sampling_ratio/mean": 0.999447762966156, "sampling/importance_sampling_ratio/max": 1.7792962789535522, "entropy": 0.4366343766450882, "clip_ratio/low_mean": 0.059206988429650664, "clip_ratio/low_min": 0.059206988429650664, "clip_ratio/high_mean": 0.03041125601157546, "clip_ratio/high_max": 0.03041125601157546, "clip_ratio/region_mean": 0.08961824444122612, "reward_total_mean": 0.45879900455474854, "reward_meter_mean": 0.27078092098236084, "reward_meter_std": 0.31442588567733765, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9928053617477417, "reward_repeat_soft_std": 0.012530048377811909, "reward_judge_quality_mean": 0.48374998569488525, "reward_judge_quality_std": 0.18965664505958557, "reward_total_composite_mean": 0.45879900455474854, "reward_total_composite_std": 0.2301139533519745} {"timestamp_utc": "2026-04-13T03:25:46Z", "mode": "train", "global_step": 2039, "epoch": 0.2048216976393772, "loss": 0.0484, "grad_norm": 16.53253746032715, "learning_rate": 3.8242424242424245e-06, "num_tokens": 3738807.0, "completions/mean_length": 26.375, "completions/min_length": 21.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.375, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.7950607538223267, "rewards/meter/std": 0.3367674648761749, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.6850000023841858, "rewards/judge_quality/std": 0.2512255907058716, "rewards/total_composite/mean": 0.8095273375511169, "rewards/total_composite/std": 0.17202194035053253, "reward": 0.8095273375511169, "reward_std": 0.17202195525169373, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0991901159286499, "sampling/sampling_logp_difference/max": 1.0164904594421387, "sampling/importance_sampling_ratio/min": 0.3618626892566681, "sampling/importance_sampling_ratio/mean": 1.0008748769760132, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7055525407195091, "clip_ratio/low_mean": 0.045517115853726864, "clip_ratio/low_min": 0.045517115853726864, "clip_ratio/high_mean": 0.07161188777536154, "clip_ratio/high_max": 0.07161188777536154, "clip_ratio/region_mean": 0.1171290036290884, "reward_total_mean": 0.8095273375511169, "reward_meter_mean": 0.7950607538223267, "reward_meter_std": 0.3367674648761749, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.6850000023841858, "reward_judge_quality_std": 0.2512255907058716, "reward_total_composite_mean": 0.8095273375511169, "reward_total_composite_std": 0.17202194035053253} {"timestamp_utc": "2026-04-13T03:25:53Z", "mode": "train", "global_step": 2040, "epoch": 0.2049221496735309, "loss": 0.0498, "grad_norm": 7.558986186981201, "learning_rate": 3.821212121212122e-06, "num_tokens": 3740930.0, "completions/mean_length": 89.375, "completions/min_length": 80.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.375, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.985985279083252, "rewards/meter/std": 0.006793915759772062, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9209914207458496, "rewards/repeat_soft/std": 0.043525390326976776, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.805417537689209, "rewards/total_composite/std": 0.02052483893930912, "reward": 0.805417537689209, "reward_std": 0.02052483707666397, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10083164274692535, "sampling/sampling_logp_difference/max": 1.3580268621444702, "sampling/importance_sampling_ratio/min": 0.2571677267551422, "sampling/importance_sampling_ratio/mean": 1.005086064338684, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5580303370952606, "clip_ratio/low_mean": 0.031810731859877706, "clip_ratio/low_min": 0.031810731859877706, "clip_ratio/high_mean": 0.08360045589506626, "clip_ratio/high_max": 0.08360045589506626, "clip_ratio/region_mean": 0.11541118775494397, "reward_total_mean": 0.805417537689209, "reward_meter_mean": 0.985985279083252, "reward_meter_std": 0.006793915759772062, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9209914207458496, "reward_repeat_soft_std": 0.043525390326976776, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.805417537689209, "reward_total_composite_std": 0.02052483893930912} {"timestamp_utc": "2026-04-13T03:26:01Z", "mode": "train", "global_step": 2041, "epoch": 0.20502260170768458, "loss": 0.015, "grad_norm": 5.526890277862549, "learning_rate": 3.818181818181819e-06, "num_tokens": 3743669.0, "completions/mean_length": 149.375, "completions/min_length": 133.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 149.375, "completions/min_terminated_length": 133.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.9795607328414917, "rewards/meter/std": 0.01771826483309269, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8807308673858643, "rewards/repeat_soft/std": 0.058018382638692856, "rewards/judge_quality/mean": 0.2799999713897705, "rewards/judge_quality/std": 0.09304375946521759, "rewards/total_composite/mean": 0.7628754377365112, "rewards/total_composite/std": 0.033323485404253006, "reward": 0.7628754377365112, "reward_std": 0.0333234928548336, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08938660472631454, "sampling/sampling_logp_difference/max": 3.119715929031372, "sampling/importance_sampling_ratio/min": 0.04416971653699875, "sampling/importance_sampling_ratio/mean": 0.9948146343231201, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4178575798869133, "clip_ratio/low_mean": 0.0532455169595778, "clip_ratio/low_min": 0.0532455169595778, "clip_ratio/high_mean": 0.03310295008122921, "clip_ratio/high_max": 0.03310295008122921, "clip_ratio/region_mean": 0.08634846704080701, "reward_total_mean": 0.7628754377365112, "reward_meter_mean": 0.9795607328414917, "reward_meter_std": 0.01771826483309269, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8807308673858643, "reward_repeat_soft_std": 0.058018382638692856, "reward_judge_quality_mean": 0.2799999713897705, "reward_judge_quality_std": 0.09304375946521759, "reward_total_composite_mean": 0.7628754377365112, "reward_total_composite_std": 0.033323485404253006} {"timestamp_utc": "2026-04-13T03:26:08Z", "mode": "train", "global_step": 2042, "epoch": 0.20512305374183828, "loss": 0.0686, "grad_norm": 16.985687255859375, "learning_rate": 3.8151515151515155e-06, "num_tokens": 3745335.0, "completions/mean_length": 39.25, "completions/min_length": 33.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.25, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.8901001214981079, "rewards/meter/std": 0.2166377454996109, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9716974496841431, "rewards/repeat_soft/std": 0.031543757766485214, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.7782148122787476, "rewards/total_composite/std": 0.0961998850107193, "reward": 0.7782148122787476, "reward_std": 0.0961998701095581, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10388285666704178, "sampling/sampling_logp_difference/max": 1.7483775615692139, "sampling/importance_sampling_ratio/min": 0.17405609786510468, "sampling/importance_sampling_ratio/mean": 1.0078333616256714, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40970541909337044, "clip_ratio/low_mean": 0.02111383154988289, "clip_ratio/low_min": 0.02111383154988289, "clip_ratio/high_mean": 0.05845301318913698, "clip_ratio/high_max": 0.05845301318913698, "clip_ratio/region_mean": 0.07956684473901987, "reward_total_mean": 0.7782148122787476, "reward_meter_mean": 0.8901001214981079, "reward_meter_std": 0.2166377454996109, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9716974496841431, "reward_repeat_soft_std": 0.031543757766485214, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.7782148122787476, "reward_total_composite_std": 0.0961998850107193} {"timestamp_utc": "2026-04-13T03:26:18Z", "mode": "train", "global_step": 2043, "epoch": 0.20522350577599197, "loss": 0.0567, "grad_norm": 8.483684539794922, "learning_rate": 3.8121212121212127e-06, "num_tokens": 3747034.0, "completions/mean_length": 58.375, "completions/min_length": 53.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.375, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.862462043762207, "rewards/meter/std": 0.21739815175533295, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8887678384780884, "rewards/repeat_soft/std": 0.05519786477088928, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.7739846706390381, "rewards/total_composite/std": 0.12426549196243286, "reward": 0.7739846706390381, "reward_std": 0.12426550686359406, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07496324181556702, "sampling/sampling_logp_difference/max": 1.2445611953735352, "sampling/importance_sampling_ratio/min": 0.2880672812461853, "sampling/importance_sampling_ratio/mean": 1.0230780839920044, "sampling/importance_sampling_ratio/max": 1.8283623456954956, "entropy": 0.3961075898259878, "clip_ratio/low_mean": 0.027276335400529206, "clip_ratio/low_min": 0.027276335400529206, "clip_ratio/high_mean": 0.04038475500419736, "clip_ratio/high_max": 0.04038475500419736, "clip_ratio/region_mean": 0.06766109040472656, "reward_total_mean": 0.7739846706390381, "reward_meter_mean": 0.862462043762207, "reward_meter_std": 0.21739815175533295, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8887678384780884, "reward_repeat_soft_std": 0.05519786477088928, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.7739846706390381, "reward_total_composite_std": 0.12426549196243286} {"timestamp_utc": "2026-04-13T03:26:25Z", "mode": "train", "global_step": 2044, "epoch": 0.20532395781014565, "loss": 0.0379, "grad_norm": 9.947355270385742, "learning_rate": 3.8090909090909095e-06, "num_tokens": 3748851.0, "completions/mean_length": 63.125, "completions/min_length": 54.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.125, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.8993710279464722, "rewards/meter/std": 0.23541277647018433, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9225738644599915, "rewards/repeat_soft/std": 0.05150970071554184, "rewards/judge_quality/mean": 0.3137499988079071, "rewards/judge_quality/std": 0.12772038578987122, "rewards/total_composite/mean": 0.7410993576049805, "rewards/total_composite/std": 0.11553937941789627, "reward": 0.7410993576049805, "reward_std": 0.11553937196731567, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10410269349813461, "sampling/sampling_logp_difference/max": 2.5679657459259033, "sampling/importance_sampling_ratio/min": 0.0766913965344429, "sampling/importance_sampling_ratio/mean": 1.0044282674789429, "sampling/importance_sampling_ratio/max": 1.8225575685501099, "entropy": 0.5503912791609764, "clip_ratio/low_mean": 0.036379615077748895, "clip_ratio/low_min": 0.036379615077748895, "clip_ratio/high_mean": 0.053158069495111704, "clip_ratio/high_max": 0.053158069495111704, "clip_ratio/region_mean": 0.0895376845728606, "reward_total_mean": 0.7410993576049805, "reward_meter_mean": 0.8993710279464722, "reward_meter_std": 0.23541277647018433, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9225738644599915, "reward_repeat_soft_std": 0.05150970071554184, "reward_judge_quality_mean": 0.3137499988079071, "reward_judge_quality_std": 0.12772038578987122, "reward_total_composite_mean": 0.7410993576049805, "reward_total_composite_std": 0.11553937941789627} {"timestamp_utc": "2026-04-13T03:26:34Z", "mode": "train", "global_step": 2045, "epoch": 0.20542440984429935, "loss": 0.0516, "grad_norm": 6.4005818367004395, "learning_rate": 3.8060606060606064e-06, "num_tokens": 3751043.0, "completions/mean_length": 112.0, "completions/min_length": 99.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.0, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.8265728950500488, "rewards/meter/std": 0.2732450067996979, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8709549903869629, "rewards/repeat_soft/std": 0.06842896342277527, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.14520922303199768, "rewards/total_composite/mean": 0.7335532903671265, "rewards/total_composite/std": 0.1336662769317627, "reward": 0.7335532903671265, "reward_std": 0.1336662769317627, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07569257915019989, "sampling/sampling_logp_difference/max": 1.3193318843841553, "sampling/importance_sampling_ratio/min": 0.2673138380050659, "sampling/importance_sampling_ratio/mean": 1.0085439682006836, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.38545671477913857, "clip_ratio/low_mean": 0.02402221830561757, "clip_ratio/low_min": 0.02402221830561757, "clip_ratio/high_mean": 0.04310059314593673, "clip_ratio/high_max": 0.04310059314593673, "clip_ratio/region_mean": 0.0671228114515543, "reward_total_mean": 0.7335532903671265, "reward_meter_mean": 0.8265728950500488, "reward_meter_std": 0.2732450067996979, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8709549903869629, "reward_repeat_soft_std": 0.06842896342277527, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.14520922303199768, "reward_total_composite_mean": 0.7335532903671265, "reward_total_composite_std": 0.1336662769317627} {"timestamp_utc": "2026-04-13T03:26:41Z", "mode": "train", "global_step": 2046, "epoch": 0.20552486187845304, "loss": -0.0177, "grad_norm": 6.545820236206055, "learning_rate": 3.803030303030303e-06, "num_tokens": 3753336.0, "completions/mean_length": 109.625, "completions/min_length": 96.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.625, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.9809578657150269, "rewards/meter/std": 0.014332360588014126, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8052318692207336, "rewards/repeat_soft/std": 0.05851438641548157, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.12351980805397034, "rewards/total_composite/mean": 0.7649542093276978, "rewards/total_composite/std": 0.03656582161784172, "reward": 0.7649542093276978, "reward_std": 0.03656582534313202, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06043347716331482, "sampling/sampling_logp_difference/max": 1.5912501811981201, "sampling/importance_sampling_ratio/min": 0.20367082953453064, "sampling/importance_sampling_ratio/mean": 1.0053848028182983, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3181152306497097, "clip_ratio/low_mean": 0.035146231297403574, "clip_ratio/low_min": 0.035146231297403574, "clip_ratio/high_mean": 0.030851224437355995, "clip_ratio/high_max": 0.030851224437355995, "clip_ratio/region_mean": 0.06599745573475957, "reward_total_mean": 0.7649542093276978, "reward_meter_mean": 0.9809578657150269, "reward_meter_std": 0.014332360588014126, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8052318692207336, "reward_repeat_soft_std": 0.05851438641548157, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.12351980805397034, "reward_total_composite_mean": 0.7649542093276978, "reward_total_composite_std": 0.03656582161784172} {"timestamp_utc": "2026-04-13T03:26:51Z", "mode": "train", "global_step": 2047, "epoch": 0.20562531391260672, "loss": -0.0033, "grad_norm": 8.854893684387207, "learning_rate": 3.8000000000000005e-06, "num_tokens": 3755137.0, "completions/mean_length": 56.125, "completions/min_length": 51.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.125, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9671645164489746, "rewards/meter/std": 0.04284930229187012, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.974216103553772, "rewards/repeat_soft/std": 0.02388555370271206, "rewards/judge_quality/mean": 0.4012500047683716, "rewards/judge_quality/std": 0.10260012745857239, "rewards/total_composite/mean": 0.8030206561088562, "rewards/total_composite/std": 0.03460470587015152, "reward": 0.8030206561088562, "reward_std": 0.03460470214486122, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09575577080249786, "sampling/sampling_logp_difference/max": 1.5687999725341797, "sampling/importance_sampling_ratio/min": 0.329790323972702, "sampling/importance_sampling_ratio/mean": 1.0102626085281372, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6411834135651588, "clip_ratio/low_mean": 0.018504902254790068, "clip_ratio/low_min": 0.018504902254790068, "clip_ratio/high_mean": 0.06197952339425683, "clip_ratio/high_max": 0.06197952339425683, "clip_ratio/region_mean": 0.0804844256490469, "reward_total_mean": 0.8030206561088562, "reward_meter_mean": 0.9671645164489746, "reward_meter_std": 0.04284930229187012, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.974216103553772, "reward_repeat_soft_std": 0.02388555370271206, "reward_judge_quality_mean": 0.4012500047683716, "reward_judge_quality_std": 0.10260012745857239, "reward_total_composite_mean": 0.8030206561088562, "reward_total_composite_std": 0.03460470587015152} {"timestamp_utc": "2026-04-13T03:27:04Z", "mode": "train", "global_step": 2048, "epoch": 0.20572576594676042, "loss": 0.0456, "grad_norm": 6.605494976043701, "learning_rate": 3.7969696969696973e-06, "num_tokens": 3757519.0, "completions/mean_length": 117.75, "completions/min_length": 106.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.75, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.7686573266983032, "rewards/meter/std": 0.20616483688354492, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9390791654586792, "rewards/repeat_soft/std": 0.02490236982703209, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.6966787576675415, "rewards/total_composite/std": 0.09288868308067322, "reward": 0.6966787576675415, "reward_std": 0.09288866817951202, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10345505177974701, "sampling/sampling_logp_difference/max": 1.9983628988265991, "sampling/importance_sampling_ratio/min": 0.13555702567100525, "sampling/importance_sampling_ratio/mean": 1.006633996963501, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5455589294433594, "clip_ratio/low_mean": 0.03795222006738186, "clip_ratio/low_min": 0.03795222006738186, "clip_ratio/high_mean": 0.06359248794615269, "clip_ratio/high_max": 0.06359248794615269, "clip_ratio/region_mean": 0.10154470801353455, "reward_total_mean": 0.6966787576675415, "reward_meter_mean": 0.7686573266983032, "reward_meter_std": 0.20616483688354492, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9390791654586792, "reward_repeat_soft_std": 0.02490236982703209, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.6966787576675415, "reward_total_composite_std": 0.09288868308067322} {"timestamp_utc": "2026-04-13T03:27:11Z", "mode": "train", "global_step": 2049, "epoch": 0.2058262179809141, "loss": 0.0634, "grad_norm": 10.388741493225098, "learning_rate": 3.793939393939394e-06, "num_tokens": 3759484.0, "completions/mean_length": 78.625, "completions/min_length": 68.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.625, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.8751641511917114, "rewards/meter/std": 0.19722811877727509, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9476295113563538, "rewards/repeat_soft/std": 0.04234865680336952, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.7390868663787842, "rewards/total_composite/std": 0.0837332084774971, "reward": 0.7390868663787842, "reward_std": 0.0837332010269165, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09820963442325592, "sampling/sampling_logp_difference/max": 1.6786552667617798, "sampling/importance_sampling_ratio/min": 0.18662476539611816, "sampling/importance_sampling_ratio/mean": 1.0141019821166992, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3993132673203945, "clip_ratio/low_mean": 0.01948866853490472, "clip_ratio/low_min": 0.01948866853490472, "clip_ratio/high_mean": 0.07153824786655605, "clip_ratio/high_max": 0.07153824786655605, "clip_ratio/region_mean": 0.09102691640146077, "reward_total_mean": 0.7390868663787842, "reward_meter_mean": 0.8751641511917114, "reward_meter_std": 0.19722811877727509, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9476295113563538, "reward_repeat_soft_std": 0.04234865680336952, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.7390868663787842, "reward_total_composite_std": 0.0837332084774971} {"timestamp_utc": "2026-04-13T03:27:18Z", "mode": "train", "global_step": 2050, "epoch": 0.20592667001506781, "loss": 0.0228, "grad_norm": 6.196682453155518, "learning_rate": 3.7909090909090914e-06, "num_tokens": 3761916.0, "completions/mean_length": 120.0, "completions/min_length": 112.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.0, "completions/min_terminated_length": 112.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.9704601764678955, "rewards/meter/std": 0.03143839165568352, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.902003288269043, "rewards/repeat_soft/std": 0.02384759485721588, "rewards/judge_quality/mean": 0.29249998927116394, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7646573781967163, "rewards/total_composite/std": 0.029248913750052452, "reward": 0.7646573781967163, "reward_std": 0.0292489193379879, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0820208191871643, "sampling/sampling_logp_difference/max": 2.2448320388793945, "sampling/importance_sampling_ratio/min": 0.10594533383846283, "sampling/importance_sampling_ratio/mean": 1.0084940195083618, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.38027220219373703, "clip_ratio/low_mean": 0.043975166510790586, "clip_ratio/low_min": 0.043975166510790586, "clip_ratio/high_mean": 0.03216180484741926, "clip_ratio/high_max": 0.03216180484741926, "clip_ratio/region_mean": 0.07613697135820985, "reward_total_mean": 0.7646573781967163, "reward_meter_mean": 0.9704601764678955, "reward_meter_std": 0.03143839165568352, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.902003288269043, "reward_repeat_soft_std": 0.02384759485721588, "reward_judge_quality_mean": 0.29249998927116394, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7646573781967163, "reward_total_composite_std": 0.029248913750052452} {"timestamp_utc": "2026-04-13T03:28:09Z", "mode": "eval", "global_step": 2050, "epoch": 0.20592667001506781, "eval_loss": NaN, "eval_runtime": 49.9321, "eval_samples_per_second": 1.602, "eval_steps_per_second": 0.2, "eval_num_tokens": 3761916.0, "eval_completions/mean_length": 99.375, "eval_completions/min_length": 38.9, "eval_completions/max_length": 194.1, "eval_completions/clipped_ratio": 0.0125, "eval_completions/mean_terminated_length": 93.94642868041993, "eval_completions/min_terminated_length": 38.9, "eval_completions/max_terminated_length": 158.9, "eval_rewards/meter/mean": 0.856502217054367, "eval_rewards/meter/std": 0.19345195293426515, "eval_rewards/count_adherence/mean": 0.9795833349227905, "eval_rewards/count_adherence/std": 0.052863120660185815, "eval_rewards/hard_gate/mean": 1.0, "eval_rewards/hard_gate/std": 0.0, "eval_rewards/repeat_soft/mean": 0.923655492067337, "eval_rewards/repeat_soft/std": 0.06058349125087261, "eval_rewards/judge_quality/mean": 0.3733749985694885, "eval_rewards/judge_quality/std": 0.1422022707760334, "eval_rewards/total_composite/mean": 0.736741554737091, "eval_rewards/total_composite/std": 0.09890033267438411, "eval_reward": 0.736741554737091, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.04979572854936123, "eval_sampling/sampling_logp_difference/max": 0.9349705219268799, "eval_sampling/importance_sampling_ratio/min": 0.3980726420879364, "eval_sampling/importance_sampling_ratio/mean": 1.0115129590034484, "eval_sampling/importance_sampling_ratio/max": 1.4258937954902648, "eval_entropy": 0.5352108478546143, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.736741554737091, "eval_reward_meter_mean": 0.856502217054367, "eval_reward_meter_std": 0.19345195293426515, "eval_reward_count_adherence_mean": 0.9795833349227905, "eval_reward_count_adherence_std": 0.052863120660185815, "eval_reward_hard_gate_mean": 1.0, "eval_reward_hard_gate_std": 0.0, "eval_reward_repeat_soft_mean": 0.923655492067337, "eval_reward_repeat_soft_std": 0.06058349125087261, "eval_reward_judge_quality_mean": 0.3733749985694885, "eval_reward_judge_quality_std": 0.1422022707760334, "eval_reward_total_composite_mean": 0.736741554737091, "eval_reward_total_composite_std": 0.09890033267438411} {"timestamp_utc": "2026-04-13T03:28:19Z", "mode": "train", "global_step": 2051, "epoch": 0.2060271220492215, "loss": -0.047, "grad_norm": 7.4440507888793945, "learning_rate": 3.7878787878787882e-06, "num_tokens": 3763683.0, "completions/mean_length": 63.875, "completions/min_length": 55.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.875, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9723237752914429, "rewards/meter/std": 0.031012145802378654, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.942585825920105, "rewards/repeat_soft/std": 0.050139158964157104, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.8190543055534363, "rewards/total_composite/std": 0.04004211723804474, "reward": 0.8190543055534363, "reward_std": 0.04004211351275444, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09692925959825516, "sampling/sampling_logp_difference/max": 1.4817352294921875, "sampling/importance_sampling_ratio/min": 0.22724303603172302, "sampling/importance_sampling_ratio/mean": 1.0047404766082764, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5477369241416454, "clip_ratio/low_mean": 0.0413682583021, "clip_ratio/low_min": 0.0413682583021, "clip_ratio/high_mean": 0.025323382578790188, "clip_ratio/high_max": 0.025323382578790188, "clip_ratio/region_mean": 0.06669164088089019, "reward_total_mean": 0.8190543055534363, "reward_meter_mean": 0.9723237752914429, "reward_meter_std": 0.031012145802378654, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.942585825920105, "reward_repeat_soft_std": 0.050139158964157104, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.8190543055534363, "reward_total_composite_std": 0.04004211723804474} {"timestamp_utc": "2026-04-13T03:28:26Z", "mode": "train", "global_step": 2052, "epoch": 0.20612757408337518, "loss": 0.0321, "grad_norm": 7.228452682495117, "learning_rate": 3.784848484848485e-06, "num_tokens": 3765966.0, "completions/mean_length": 112.375, "completions/min_length": 99.0, "completions/max_length": 119.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.375, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.9267852306365967, "rewards/meter/std": 0.09602653235197067, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9362842440605164, "rewards/repeat_soft/std": 0.03436482325196266, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7803068161010742, "rewards/total_composite/std": 0.04212849587202072, "reward": 0.7803068161010742, "reward_std": 0.042128488421440125, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09706283360719681, "sampling/sampling_logp_difference/max": 1.7946081161499023, "sampling/importance_sampling_ratio/min": 0.16619257628917694, "sampling/importance_sampling_ratio/mean": 1.004345417022705, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5432569235563278, "clip_ratio/low_mean": 0.0356596102938056, "clip_ratio/low_min": 0.0356596102938056, "clip_ratio/high_mean": 0.06272720452398062, "clip_ratio/high_max": 0.06272720452398062, "clip_ratio/region_mean": 0.09838681481778622, "reward_total_mean": 0.7803068161010742, "reward_meter_mean": 0.9267852306365967, "reward_meter_std": 0.09602653235197067, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9362842440605164, "reward_repeat_soft_std": 0.03436482325196266, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7803068161010742, "reward_total_composite_std": 0.04212849587202072} {"timestamp_utc": "2026-04-13T03:28:32Z", "mode": "train", "global_step": 2053, "epoch": 0.20622802611752888, "loss": 0.0612, "grad_norm": 21.70195770263672, "learning_rate": 3.781818181818182e-06, "num_tokens": 3767440.0, "completions/mean_length": 28.25, "completions/min_length": 24.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.25, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9790338277816772, "rewards/meter/std": 0.019558291882276535, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9567934274673462, "rewards/repeat_soft/std": 0.016140474006533623, "rewards/judge_quality/mean": 0.9200000166893005, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.9622445702552795, "rewards/total_composite/std": 0.008447692729532719, "reward": 0.9622445702552795, "reward_std": 0.008447702042758465, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1262401044368744, "sampling/sampling_logp_difference/max": 1.073310375213623, "sampling/importance_sampling_ratio/min": 0.341874897480011, "sampling/importance_sampling_ratio/mean": 1.0202358961105347, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7634189799427986, "clip_ratio/low_mean": 0.06371941231191158, "clip_ratio/low_min": 0.06371941231191158, "clip_ratio/high_mean": 0.07164152432233095, "clip_ratio/high_max": 0.07164152432233095, "clip_ratio/region_mean": 0.13536093663424253, "reward_total_mean": 0.9622445702552795, "reward_meter_mean": 0.9790338277816772, "reward_meter_std": 0.019558291882276535, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9567934274673462, "reward_repeat_soft_std": 0.016140474006533623, "reward_judge_quality_mean": 0.9200000166893005, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.9622445702552795, "reward_total_composite_std": 0.008447692729532719} {"timestamp_utc": "2026-04-13T03:28:38Z", "mode": "train", "global_step": 2054, "epoch": 0.20632847815168257, "loss": 0.1034, "grad_norm": 8.019661903381348, "learning_rate": 3.778787878787879e-06, "num_tokens": 3769246.0, "completions/mean_length": 77.75, "completions/min_length": 62.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.75, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.979961097240448, "rewards/meter/std": 0.013457313179969788, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.98641037940979, "rewards/repeat_soft/std": 0.013435971923172474, "rewards/judge_quality/mean": 0.44749999046325684, "rewards/judge_quality/std": 0.12848013639450073, "rewards/total_composite/mean": 0.8238735198974609, "rewards/total_composite/std": 0.04209812358021736, "reward": 0.8238735198974609, "reward_std": 0.04209813475608826, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13944080471992493, "sampling/sampling_logp_difference/max": 1.5801262855529785, "sampling/importance_sampling_ratio/min": 0.2059490829706192, "sampling/importance_sampling_ratio/mean": 1.0181822776794434, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9059241563081741, "clip_ratio/low_mean": 0.07420717226341367, "clip_ratio/low_min": 0.07420717226341367, "clip_ratio/high_mean": 0.03514863969758153, "clip_ratio/high_max": 0.03514863969758153, "clip_ratio/region_mean": 0.1093558119609952, "reward_total_mean": 0.8238735198974609, "reward_meter_mean": 0.979961097240448, "reward_meter_std": 0.013457313179969788, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.98641037940979, "reward_repeat_soft_std": 0.013435971923172474, "reward_judge_quality_mean": 0.44749999046325684, "reward_judge_quality_std": 0.12848013639450073, "reward_total_composite_mean": 0.8238735198974609, "reward_total_composite_std": 0.04209812358021736} {"timestamp_utc": "2026-04-13T03:28:46Z", "mode": "train", "global_step": 2055, "epoch": 0.20642893018583627, "loss": 0.0386, "grad_norm": 6.990056037902832, "learning_rate": 3.775757575757576e-06, "num_tokens": 3771529.0, "completions/mean_length": 112.375, "completions/min_length": 106.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.375, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.8467268347740173, "rewards/meter/std": 0.2065541297197342, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9679893255233765, "rewards/repeat_soft/std": 0.010920184664428234, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7538260221481323, "rewards/total_composite/std": 0.0935600996017456, "reward": 0.7538260221481323, "reward_std": 0.0935600996017456, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08980797231197357, "sampling/sampling_logp_difference/max": 1.5732793807983398, "sampling/importance_sampling_ratio/min": 0.2073640376329422, "sampling/importance_sampling_ratio/mean": 0.9967440962791443, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.35258666053414345, "clip_ratio/low_mean": 0.026839465368539095, "clip_ratio/low_min": 0.026839465368539095, "clip_ratio/high_mean": 0.06027027452364564, "clip_ratio/high_max": 0.06027027452364564, "clip_ratio/region_mean": 0.08710973989218473, "reward_total_mean": 0.7538260221481323, "reward_meter_mean": 0.8467268347740173, "reward_meter_std": 0.2065541297197342, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9679893255233765, "reward_repeat_soft_std": 0.010920184664428234, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7538260221481323, "reward_total_composite_std": 0.0935600996017456} {"timestamp_utc": "2026-04-13T03:28:57Z", "mode": "train", "global_step": 2056, "epoch": 0.20652938221998995, "loss": -0.2461, "grad_norm": 2.064955234527588, "learning_rate": 3.772727272727273e-06, "num_tokens": 3774436.0, "completions/mean_length": 216.375, "completions/min_length": 151.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 174.1428680419922, "completions/min_terminated_length": 151.0, "completions/max_terminated_length": 196.0, "rewards/meter/mean": 0.9707460403442383, "rewards/meter/std": 0.031688038259744644, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9348890781402588, "rewards/repeat_soft/std": 0.04012514650821686, "rewards/judge_quality/mean": 0.29749998450279236, "rewards/judge_quality/std": 0.14518460631370544, "rewards/total_composite/mean": 0.6777241230010986, "rewards/total_composite/std": 0.27647343277931213, "reward": 0.6777241230010986, "reward_std": 0.27647343277931213, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12337951362133026, "sampling/sampling_logp_difference/max": 2.0074384212493896, "sampling/importance_sampling_ratio/min": 0.1343323439359665, "sampling/importance_sampling_ratio/mean": 0.9996081590652466, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6158844456076622, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09989127423614264, "clip_ratio/high_max": 0.09989127423614264, "clip_ratio/region_mean": 0.09989127423614264, "reward_total_mean": 0.6777241230010986, "reward_meter_mean": 0.9707460403442383, "reward_meter_std": 0.031688038259744644, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9348890781402588, "reward_repeat_soft_std": 0.04012514650821686, "reward_judge_quality_mean": 0.29749998450279236, "reward_judge_quality_std": 0.14518460631370544, "reward_total_composite_mean": 0.6777241230010986, "reward_total_composite_std": 0.27647343277931213} {"timestamp_utc": "2026-04-13T03:29:04Z", "mode": "train", "global_step": 2057, "epoch": 0.20662983425414364, "loss": -0.0197, "grad_norm": 8.446084022521973, "learning_rate": 3.76969696969697e-06, "num_tokens": 3776128.0, "completions/mean_length": 58.5, "completions/min_length": 53.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.5, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9601409435272217, "rewards/meter/std": 0.04742641746997833, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9566521644592285, "rewards/repeat_soft/std": 0.029788490384817123, "rewards/judge_quality/mean": 0.5900000333786011, "rewards/judge_quality/std": 0.27994900941848755, "rewards/total_composite/mean": 0.854728639125824, "rewards/total_composite/std": 0.07888007909059525, "reward": 0.854728639125824, "reward_std": 0.07888009399175644, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08524399995803833, "sampling/sampling_logp_difference/max": 1.5382575988769531, "sampling/importance_sampling_ratio/min": 0.21475496888160706, "sampling/importance_sampling_ratio/mean": 0.997784435749054, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4865318648517132, "clip_ratio/low_mean": 0.03334306995384395, "clip_ratio/low_min": 0.03334306995384395, "clip_ratio/high_mean": 0.029897186439484358, "clip_ratio/high_max": 0.029897186439484358, "clip_ratio/region_mean": 0.06324025639332831, "reward_total_mean": 0.854728639125824, "reward_meter_mean": 0.9601409435272217, "reward_meter_std": 0.04742641746997833, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9566521644592285, "reward_repeat_soft_std": 0.029788490384817123, "reward_judge_quality_mean": 0.5900000333786011, "reward_judge_quality_std": 0.27994900941848755, "reward_total_composite_mean": 0.854728639125824, "reward_total_composite_std": 0.07888007909059525} {"timestamp_utc": "2026-04-13T03:29:11Z", "mode": "train", "global_step": 2058, "epoch": 0.20673028628829734, "loss": 0.0678, "grad_norm": 18.829803466796875, "learning_rate": 3.766666666666667e-06, "num_tokens": 3777714.0, "completions/mean_length": 31.25, "completions/min_length": 25.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.25, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.7327589988708496, "rewards/meter/std": 0.35304829478263855, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9562379121780396, "rewards/repeat_soft/std": 0.013668385334312916, "rewards/judge_quality/mean": 0.5850000381469727, "rewards/judge_quality/std": 0.2946668863296509, "rewards/total_composite/mean": 0.7508653402328491, "rewards/total_composite/std": 0.2154407501220703, "reward": 0.7508653402328491, "reward_std": 0.21544072031974792, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1484455019235611, "sampling/sampling_logp_difference/max": 1.1443462371826172, "sampling/importance_sampling_ratio/min": 0.3184320330619812, "sampling/importance_sampling_ratio/mean": 1.0296164751052856, "sampling/importance_sampling_ratio/max": 1.908284068107605, "entropy": 0.873430609703064, "clip_ratio/low_mean": 0.04125000070780516, "clip_ratio/low_min": 0.04125000070780516, "clip_ratio/high_mean": 0.0892522456124425, "clip_ratio/high_max": 0.0892522456124425, "clip_ratio/region_mean": 0.13050224632024765, "reward_total_mean": 0.7508653402328491, "reward_meter_mean": 0.7327589988708496, "reward_meter_std": 0.35304829478263855, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9562379121780396, "reward_repeat_soft_std": 0.013668385334312916, "reward_judge_quality_mean": 0.5850000381469727, "reward_judge_quality_std": 0.2946668863296509, "reward_total_composite_mean": 0.7508653402328491, "reward_total_composite_std": 0.2154407501220703} {"timestamp_utc": "2026-04-13T03:29:17Z", "mode": "train", "global_step": 2059, "epoch": 0.20683073832245102, "loss": 0.132, "grad_norm": 12.751532554626465, "learning_rate": 3.7636363636363637e-06, "num_tokens": 3779636.0, "completions/mean_length": 57.25, "completions/min_length": 41.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.25, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.7533246278762817, "rewards/meter/std": 0.36882948875427246, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9769197702407837, "rewards/repeat_soft/std": 0.01318880170583725, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.7096880674362183, "rewards/total_composite/std": 0.16908150911331177, "reward": 0.7096880674362183, "reward_std": 0.16908149421215057, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12830428779125214, "sampling/sampling_logp_difference/max": 1.360590934753418, "sampling/importance_sampling_ratio/min": 0.2565091550350189, "sampling/importance_sampling_ratio/mean": 0.9919193983078003, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7646838650107384, "clip_ratio/low_mean": 0.030705128330737352, "clip_ratio/low_min": 0.030705128330737352, "clip_ratio/high_mean": 0.11295096017420292, "clip_ratio/high_max": 0.11295096017420292, "clip_ratio/region_mean": 0.14365608850494027, "reward_total_mean": 0.7096880674362183, "reward_meter_mean": 0.7533246278762817, "reward_meter_std": 0.36882948875427246, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9769197702407837, "reward_repeat_soft_std": 0.01318880170583725, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.7096880674362183, "reward_total_composite_std": 0.16908150911331177} {"timestamp_utc": "2026-04-13T03:29:24Z", "mode": "train", "global_step": 2060, "epoch": 0.20693119035660473, "loss": 0.0447, "grad_norm": 14.643370628356934, "learning_rate": 3.7606060606060605e-06, "num_tokens": 3781243.0, "completions/mean_length": 39.875, "completions/min_length": 35.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.875, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.29759132862091064, "rewards/meter/std": 0.2520938217639923, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9754296541213989, "rewards/repeat_soft/std": 0.021340761333703995, "rewards/judge_quality/mean": 0.7362500429153442, "rewards/judge_quality/std": 0.25376805663108826, "rewards/total_composite/mean": 0.6023340225219727, "rewards/total_composite/std": 0.10581181198358536, "reward": 0.6023340225219727, "reward_std": 0.10581182688474655, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12221232801675797, "sampling/sampling_logp_difference/max": 1.2984979152679443, "sampling/importance_sampling_ratio/min": 0.2729414701461792, "sampling/importance_sampling_ratio/mean": 0.9833239912986755, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6576742604374886, "clip_ratio/low_mean": 0.04157758550718427, "clip_ratio/low_min": 0.04157758550718427, "clip_ratio/high_mean": 0.07668538112193346, "clip_ratio/high_max": 0.07668538112193346, "clip_ratio/region_mean": 0.11826296662911773, "reward_total_mean": 0.6023340225219727, "reward_meter_mean": 0.29759132862091064, "reward_meter_std": 0.2520938217639923, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9754296541213989, "reward_repeat_soft_std": 0.021340761333703995, "reward_judge_quality_mean": 0.7362500429153442, "reward_judge_quality_std": 0.25376805663108826, "reward_total_composite_mean": 0.6023340225219727, "reward_total_composite_std": 0.10581181198358536} {"timestamp_utc": "2026-04-13T03:29:30Z", "mode": "train", "global_step": 2061, "epoch": 0.2070316423907584, "loss": 0.1347, "grad_norm": 10.4701566696167, "learning_rate": 3.757575757575758e-06, "num_tokens": 3783127.0, "completions/mean_length": 67.5, "completions/min_length": 51.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.5, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.8245259523391724, "rewards/meter/std": 0.21519596874713898, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9871628284454346, "rewards/repeat_soft/std": 0.014667271636426449, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7513779401779175, "rewards/total_composite/std": 0.0967240259051323, "reward": 0.7513779401779175, "reward_std": 0.09672403335571289, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13258886337280273, "sampling/sampling_logp_difference/max": 2.0349230766296387, "sampling/importance_sampling_ratio/min": 0.1306905448436737, "sampling/importance_sampling_ratio/mean": 1.013372778892517, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.829378604888916, "clip_ratio/low_mean": 0.05312240496277809, "clip_ratio/low_min": 0.05312240496277809, "clip_ratio/high_mean": 0.09418727643787861, "clip_ratio/high_max": 0.09418727643787861, "clip_ratio/region_mean": 0.1473096814006567, "reward_total_mean": 0.7513779401779175, "reward_meter_mean": 0.8245259523391724, "reward_meter_std": 0.21519596874713898, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9871628284454346, "reward_repeat_soft_std": 0.014667271636426449, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7513779401779175, "reward_total_composite_std": 0.0967240259051323} {"timestamp_utc": "2026-04-13T03:29:42Z", "mode": "train", "global_step": 2062, "epoch": 0.2071320944249121, "loss": -0.0752, "grad_norm": 4.111029148101807, "learning_rate": 3.7545454545454546e-06, "num_tokens": 3784570.0, "completions/mean_length": 92.375, "completions/min_length": 28.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 32.42857360839844, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.46550846099853516, "rewards/meter/std": 0.4284335970878601, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.5174999833106995, "rewards/judge_quality/std": 0.2841905951499939, "rewards/total_composite/mean": 0.577812671661377, "rewards/total_composite/std": 0.30195993185043335, "reward": 0.577812671661377, "reward_std": 0.30195993185043335, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1547636091709137, "sampling/sampling_logp_difference/max": 1.70302152633667, "sampling/importance_sampling_ratio/min": 0.18213237822055817, "sampling/importance_sampling_ratio/mean": 1.0080779790878296, "sampling/importance_sampling_ratio/max": 1.8609474897384644, "entropy": 0.6802561096847057, "clip_ratio/low_mean": 0.0251736119389534, "clip_ratio/low_min": 0.0251736119389534, "clip_ratio/high_mean": 0.07414459809660912, "clip_ratio/high_max": 0.07414459809660912, "clip_ratio/region_mean": 0.09931821003556252, "reward_total_mean": 0.577812671661377, "reward_meter_mean": 0.46550846099853516, "reward_meter_std": 0.4284335970878601, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.5174999833106995, "reward_judge_quality_std": 0.2841905951499939, "reward_total_composite_mean": 0.577812671661377, "reward_total_composite_std": 0.30195993185043335} {"timestamp_utc": "2026-04-13T03:29:50Z", "mode": "train", "global_step": 2063, "epoch": 0.2072325464590658, "loss": 0.0251, "grad_norm": 7.1173810958862305, "learning_rate": 3.7515151515151515e-06, "num_tokens": 3787026.0, "completions/mean_length": 135.0, "completions/min_length": 114.0, "completions/max_length": 155.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 135.0, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 155.0, "rewards/meter/mean": 0.8920982480049133, "rewards/meter/std": 0.0885041132569313, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9669294357299805, "rewards/repeat_soft/std": 0.019839514046907425, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7677621841430664, "rewards/total_composite/std": 0.03599567338824272, "reward": 0.7677621841430664, "reward_std": 0.03599567338824272, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1087738499045372, "sampling/sampling_logp_difference/max": 1.882472276687622, "sampling/importance_sampling_ratio/min": 0.15221333503723145, "sampling/importance_sampling_ratio/mean": 1.014733910560608, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6673064976930618, "clip_ratio/low_mean": 0.06245468929409981, "clip_ratio/low_min": 0.06245468929409981, "clip_ratio/high_mean": 0.039204804226756096, "clip_ratio/high_max": 0.039204804226756096, "clip_ratio/region_mean": 0.1016594935208559, "reward_total_mean": 0.7677621841430664, "reward_meter_mean": 0.8920982480049133, "reward_meter_std": 0.0885041132569313, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9669294357299805, "reward_repeat_soft_std": 0.019839514046907425, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7677621841430664, "reward_total_composite_std": 0.03599567338824272} {"timestamp_utc": "2026-04-13T03:29:56Z", "mode": "train", "global_step": 2064, "epoch": 0.20733299849321948, "loss": 0.0545, "grad_norm": 19.70221519470215, "learning_rate": 3.748484848484849e-06, "num_tokens": 3788481.0, "completions/mean_length": 26.875, "completions/min_length": 21.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.875, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.7770399451255798, "rewards/meter/std": 0.2687740623950958, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9587112665176392, "rewards/repeat_soft/std": 0.01037901733070612, "rewards/judge_quality/mean": 0.2762500047683716, "rewards/judge_quality/std": 0.12603145837783813, "rewards/total_composite/mean": 0.6596641540527344, "rewards/total_composite/std": 0.09364897757768631, "reward": 0.6596641540527344, "reward_std": 0.09364897012710571, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11279717087745667, "sampling/sampling_logp_difference/max": 1.131174087524414, "sampling/importance_sampling_ratio/min": 0.32265421748161316, "sampling/importance_sampling_ratio/mean": 1.0084378719329834, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5971592888236046, "clip_ratio/low_mean": 0.02846790896728635, "clip_ratio/low_min": 0.02846790896728635, "clip_ratio/high_mean": 0.059659091755747795, "clip_ratio/high_max": 0.059659091755747795, "clip_ratio/region_mean": 0.08812700072303414, "reward_total_mean": 0.6596641540527344, "reward_meter_mean": 0.7770399451255798, "reward_meter_std": 0.2687740623950958, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9587112665176392, "reward_repeat_soft_std": 0.01037901733070612, "reward_judge_quality_mean": 0.2762500047683716, "reward_judge_quality_std": 0.12603145837783813, "reward_total_composite_mean": 0.6596641540527344, "reward_total_composite_std": 0.09364897757768631} {"timestamp_utc": "2026-04-13T03:30:03Z", "mode": "train", "global_step": 2065, "epoch": 0.2074334505273732, "loss": -0.0024, "grad_norm": 9.843429565429688, "learning_rate": 3.745454545454546e-06, "num_tokens": 3790123.0, "completions/mean_length": 37.25, "completions/min_length": 32.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.25, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.6114059090614319, "rewards/meter/std": 0.23902292549610138, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9875791072845459, "rewards/repeat_soft/std": 0.01578284241259098, "rewards/judge_quality/mean": 0.4674999713897705, "rewards/judge_quality/std": 0.07497618347406387, "rewards/total_composite/mean": 0.6641405820846558, "rewards/total_composite/std": 0.10383687913417816, "reward": 0.6641405820846558, "reward_std": 0.10383687913417816, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13429446518421173, "sampling/sampling_logp_difference/max": 1.3016719818115234, "sampling/importance_sampling_ratio/min": 0.2720765173435211, "sampling/importance_sampling_ratio/mean": 1.0106227397918701, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5150891281664371, "clip_ratio/low_mean": 0.04513888992369175, "clip_ratio/low_min": 0.04513888992369175, "clip_ratio/high_mean": 0.09191238693892956, "clip_ratio/high_max": 0.09191238693892956, "clip_ratio/region_mean": 0.1370512768626213, "reward_total_mean": 0.6641405820846558, "reward_meter_mean": 0.6114059090614319, "reward_meter_std": 0.23902292549610138, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9875791072845459, "reward_repeat_soft_std": 0.01578284241259098, "reward_judge_quality_mean": 0.4674999713897705, "reward_judge_quality_std": 0.07497618347406387, "reward_total_composite_mean": 0.6641405820846558, "reward_total_composite_std": 0.10383687913417816} {"timestamp_utc": "2026-04-13T03:30:08Z", "mode": "train", "global_step": 2066, "epoch": 0.20753390256152687, "loss": 0.0621, "grad_norm": 23.256099700927734, "learning_rate": 3.742424242424243e-06, "num_tokens": 3791674.0, "completions/mean_length": 27.875, "completions/min_length": 21.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.875, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.7411433458328247, "rewards/meter/std": 0.41209810972213745, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9615821838378906, "rewards/repeat_soft/std": 0.002596014179289341, "rewards/judge_quality/mean": 0.44999998807907104, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7146726846694946, "rewards/total_composite/std": 0.18559718132019043, "reward": 0.7146726846694946, "reward_std": 0.18559716641902924, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13531652092933655, "sampling/sampling_logp_difference/max": 1.4647266864776611, "sampling/importance_sampling_ratio/min": 0.23114116489887238, "sampling/importance_sampling_ratio/mean": 0.9989341497421265, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7549142017960548, "clip_ratio/low_mean": 0.024739583488553762, "clip_ratio/low_min": 0.024739583488553762, "clip_ratio/high_mean": 0.07615297799929976, "clip_ratio/high_max": 0.07615297799929976, "clip_ratio/region_mean": 0.10089256148785353, "reward_total_mean": 0.7146726846694946, "reward_meter_mean": 0.7411433458328247, "reward_meter_std": 0.41209810972213745, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9615821838378906, "reward_repeat_soft_std": 0.002596014179289341, "reward_judge_quality_mean": 0.44999998807907104, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7146726846694946, "reward_total_composite_std": 0.18559718132019043} {"timestamp_utc": "2026-04-13T03:30:19Z", "mode": "train", "global_step": 2067, "epoch": 0.20763435459568055, "loss": -0.1314, "grad_norm": 2.058121919631958, "learning_rate": 3.73939393939394e-06, "num_tokens": 3793132.0, "completions/mean_length": 101.25, "completions/min_length": 37.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 42.57143020629883, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.7879319787025452, "rewards/meter/std": 0.15647803246974945, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9920636415481567, "rewards/repeat_soft/std": 0.014280635863542557, "rewards/judge_quality/mean": 0.8112499713897705, "rewards/judge_quality/std": 0.3075914680957794, "rewards/total_composite/mean": 0.8471508026123047, "rewards/total_composite/std": 0.1498747617006302, "reward": 0.8471508026123047, "reward_std": 0.1498747617006302, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10783009976148605, "sampling/sampling_logp_difference/max": 1.5714166164398193, "sampling/importance_sampling_ratio/min": 0.23148736357688904, "sampling/importance_sampling_ratio/mean": 1.0107582807540894, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.41116226837038994, "clip_ratio/low_mean": 0.0099618851672858, "clip_ratio/low_min": 0.0099618851672858, "clip_ratio/high_mean": 0.05205012857913971, "clip_ratio/high_max": 0.05205012857913971, "clip_ratio/region_mean": 0.06201201374642551, "reward_total_mean": 0.8471508026123047, "reward_meter_mean": 0.7879319787025452, "reward_meter_std": 0.15647803246974945, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9920636415481567, "reward_repeat_soft_std": 0.014280635863542557, "reward_judge_quality_mean": 0.8112499713897705, "reward_judge_quality_std": 0.3075914680957794, "reward_total_composite_mean": 0.8471508026123047, "reward_total_composite_std": 0.1498747617006302} {"timestamp_utc": "2026-04-13T03:30:32Z", "mode": "train", "global_step": 2068, "epoch": 0.20773480662983426, "loss": -0.0016, "grad_norm": 8.137001991271973, "learning_rate": 3.736363636363637e-06, "num_tokens": 3795449.0, "completions/mean_length": 116.625, "completions/min_length": 98.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.625, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.47095316648483276, "rewards/meter/std": 0.3366110026836395, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.956960141658783, "rewards/repeat_soft/std": 0.0246689785271883, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.6173748970031738, "rewards/total_composite/std": 0.16809752583503723, "reward": 0.6173748970031738, "reward_std": 0.16809752583503723, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12424919009208679, "sampling/sampling_logp_difference/max": 2.311819314956665, "sampling/importance_sampling_ratio/min": 0.09908083081245422, "sampling/importance_sampling_ratio/mean": 1.0165514945983887, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7932340651750565, "clip_ratio/low_mean": 0.0785398967564106, "clip_ratio/low_min": 0.0785398967564106, "clip_ratio/high_mean": 0.04322816710919142, "clip_ratio/high_max": 0.04322816710919142, "clip_ratio/region_mean": 0.12176806386560202, "reward_total_mean": 0.6173748970031738, "reward_meter_mean": 0.47095316648483276, "reward_meter_std": 0.3366110026836395, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.956960141658783, "reward_repeat_soft_std": 0.0246689785271883, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.6173748970031738, "reward_total_composite_std": 0.16809752583503723} {"timestamp_utc": "2026-04-13T03:30:39Z", "mode": "train", "global_step": 2069, "epoch": 0.20783525866398794, "loss": 0.0407, "grad_norm": 9.756993293762207, "learning_rate": 3.7333333333333337e-06, "num_tokens": 3797204.0, "completions/mean_length": 61.375, "completions/min_length": 57.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.375, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9464276432991028, "rewards/meter/std": 0.06895659863948822, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9323275089263916, "rewards/repeat_soft/std": 0.030339930206537247, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.1865811049938202, "rewards/total_composite/mean": 0.8285001516342163, "rewards/total_composite/std": 0.06655344367027283, "reward": 0.8285001516342163, "reward_std": 0.06655344367027283, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11313411593437195, "sampling/sampling_logp_difference/max": 1.3596301078796387, "sampling/importance_sampling_ratio/min": 0.2567557394504547, "sampling/importance_sampling_ratio/mean": 1.010406255722046, "sampling/importance_sampling_ratio/max": 1.7923930883407593, "entropy": 0.6240393593907356, "clip_ratio/low_mean": 0.07132996898144484, "clip_ratio/low_min": 0.07132996898144484, "clip_ratio/high_mean": 0.02644723281264305, "clip_ratio/high_max": 0.02644723281264305, "clip_ratio/region_mean": 0.09777720179408789, "reward_total_mean": 0.8285001516342163, "reward_meter_mean": 0.9464276432991028, "reward_meter_std": 0.06895659863948822, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9323275089263916, "reward_repeat_soft_std": 0.030339930206537247, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.1865811049938202, "reward_total_composite_mean": 0.8285001516342163, "reward_total_composite_std": 0.06655344367027283} {"timestamp_utc": "2026-04-13T03:30:50Z", "mode": "train", "global_step": 2070, "epoch": 0.20793571069814162, "loss": -0.1341, "grad_norm": 2.537036895751953, "learning_rate": 3.7303030303030306e-06, "num_tokens": 3798887.0, "completions/mean_length": 112.375, "completions/min_length": 48.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 55.28571701049805, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.7053797245025635, "rewards/meter/std": 0.4090324342250824, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.958525538444519, "rewards/repeat_soft/std": 0.031624678522348404, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.13617216050624847, "rewards/total_composite/mean": 0.6456411480903625, "rewards/total_composite/std": 0.29118043184280396, "reward": 0.6456411480903625, "reward_std": 0.29118043184280396, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1050063893198967, "sampling/sampling_logp_difference/max": 1.252467155456543, "sampling/importance_sampling_ratio/min": 0.2857988178730011, "sampling/importance_sampling_ratio/mean": 1.014007806777954, "sampling/importance_sampling_ratio/max": 1.8907262086868286, "entropy": 0.6651944518089294, "clip_ratio/low_mean": 0.0022321429569274187, "clip_ratio/low_min": 0.0022321429569274187, "clip_ratio/high_mean": 0.07731214538216591, "clip_ratio/high_max": 0.07731214538216591, "clip_ratio/region_mean": 0.07954428833909333, "reward_total_mean": 0.6456411480903625, "reward_meter_mean": 0.7053797245025635, "reward_meter_std": 0.4090324342250824, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.958525538444519, "reward_repeat_soft_std": 0.031624678522348404, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.13617216050624847, "reward_total_composite_mean": 0.6456411480903625, "reward_total_composite_std": 0.29118043184280396} {"timestamp_utc": "2026-04-13T03:30:56Z", "mode": "train", "global_step": 2071, "epoch": 0.20803616273229533, "loss": 0.0447, "grad_norm": 9.124361991882324, "learning_rate": 3.727272727272728e-06, "num_tokens": 3800562.0, "completions/mean_length": 54.375, "completions/min_length": 47.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.375, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9702175855636597, "rewards/meter/std": 0.026757318526506424, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.964919924736023, "rewards/repeat_soft/std": 0.03421337530016899, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8113398551940918, "rewards/total_composite/std": 0.014529167674481869, "reward": 0.8113398551940918, "reward_std": 0.014529158361256123, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12845925986766815, "sampling/sampling_logp_difference/max": 1.8049001693725586, "sampling/importance_sampling_ratio/min": 0.16449087858200073, "sampling/importance_sampling_ratio/mean": 0.9948961734771729, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7075825482606888, "clip_ratio/low_mean": 0.04901030287146568, "clip_ratio/low_min": 0.04901030287146568, "clip_ratio/high_mean": 0.07839901838451624, "clip_ratio/high_max": 0.07839901838451624, "clip_ratio/region_mean": 0.12740932125598192, "reward_total_mean": 0.8113398551940918, "reward_meter_mean": 0.9702175855636597, "reward_meter_std": 0.026757318526506424, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.964919924736023, "reward_repeat_soft_std": 0.03421337530016899, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8113398551940918, "reward_total_composite_std": 0.014529167674481869} {"timestamp_utc": "2026-04-13T03:31:03Z", "mode": "train", "global_step": 2072, "epoch": 0.208136614766449, "loss": 0.0488, "grad_norm": 5.833095073699951, "learning_rate": 3.7242424242424246e-06, "num_tokens": 3802606.0, "completions/mean_length": 104.5, "completions/min_length": 90.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.5, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.97432941198349, "rewards/meter/std": 0.026204582303762436, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9714348316192627, "rewards/repeat_soft/std": 0.021710382774472237, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.8415917158126831, "rewards/total_composite/std": 0.06301872432231903, "reward": 0.8415917158126831, "reward_std": 0.06301873177289963, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11340118199586868, "sampling/sampling_logp_difference/max": 1.1494512557983398, "sampling/importance_sampling_ratio/min": 0.31681057810783386, "sampling/importance_sampling_ratio/mean": 1.015483021736145, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8444170355796814, "clip_ratio/low_mean": 0.08798154024407268, "clip_ratio/low_min": 0.08798154024407268, "clip_ratio/high_mean": 0.03365695849061012, "clip_ratio/high_max": 0.03365695849061012, "clip_ratio/region_mean": 0.1216384987346828, "reward_total_mean": 0.8415917158126831, "reward_meter_mean": 0.97432941198349, "reward_meter_std": 0.026204582303762436, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9714348316192627, "reward_repeat_soft_std": 0.021710382774472237, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.8415917158126831, "reward_total_composite_std": 0.06301872432231903} {"timestamp_utc": "2026-04-13T03:31:11Z", "mode": "train", "global_step": 2073, "epoch": 0.20823706680060272, "loss": -0.0225, "grad_norm": 16.089406967163086, "learning_rate": 3.7212121212121215e-06, "num_tokens": 3804228.0, "completions/mean_length": 41.75, "completions/min_length": 34.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.75, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.7821932435035706, "rewards/meter/std": 0.2383231818675995, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.945978581905365, "rewards/repeat_soft/std": 0.04304986074566841, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7237098217010498, "rewards/total_composite/std": 0.1104959025979042, "reward": 0.7237098217010498, "reward_std": 0.11049588024616241, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1236070916056633, "sampling/sampling_logp_difference/max": 2.009012222290039, "sampling/importance_sampling_ratio/min": 0.1341210901737213, "sampling/importance_sampling_ratio/mean": 1.004361867904663, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.46744443476200104, "clip_ratio/low_mean": 0.03073152434080839, "clip_ratio/low_min": 0.03073152434080839, "clip_ratio/high_mean": 0.08511652052402496, "clip_ratio/high_max": 0.08511652052402496, "clip_ratio/region_mean": 0.11584804486483335, "reward_total_mean": 0.7237098217010498, "reward_meter_mean": 0.7821932435035706, "reward_meter_std": 0.2383231818675995, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.945978581905365, "reward_repeat_soft_std": 0.04304986074566841, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7237098217010498, "reward_total_composite_std": 0.1104959025979042} {"timestamp_utc": "2026-04-13T03:31:19Z", "mode": "train", "global_step": 2074, "epoch": 0.2083375188347564, "loss": 0.055, "grad_norm": 4.822023868560791, "learning_rate": 3.7181818181818187e-06, "num_tokens": 3807356.0, "completions/mean_length": 210.0, "completions/min_length": 191.0, "completions/max_length": 234.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 210.0, "completions/min_terminated_length": 191.0, "completions/max_terminated_length": 234.0, "rewards/meter/mean": 0.968381404876709, "rewards/meter/std": 0.04713902622461319, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.838223397731781, "rewards/repeat_soft/std": 0.06866166740655899, "rewards/judge_quality/mean": 0.20000000298023224, "rewards/judge_quality/std": 0.05345224589109421, "rewards/total_composite/mean": 0.7295939922332764, "rewards/total_composite/std": 0.028615713119506836, "reward": 0.7295939922332764, "reward_std": 0.02861572429537773, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.092617928981781, "sampling/sampling_logp_difference/max": 2.070281982421875, "sampling/importance_sampling_ratio/min": 0.1261502057313919, "sampling/importance_sampling_ratio/mean": 1.0209999084472656, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5908392295241356, "clip_ratio/low_mean": 0.042885797098279, "clip_ratio/low_min": 0.042885797098279, "clip_ratio/high_mean": 0.04003082774579525, "clip_ratio/high_max": 0.04003082774579525, "clip_ratio/region_mean": 0.08291662484407425, "reward_total_mean": 0.7295939922332764, "reward_meter_mean": 0.968381404876709, "reward_meter_std": 0.04713902622461319, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.838223397731781, "reward_repeat_soft_std": 0.06866166740655899, "reward_judge_quality_mean": 0.20000000298023224, "reward_judge_quality_std": 0.05345224589109421, "reward_total_composite_mean": 0.7295939922332764, "reward_total_composite_std": 0.028615713119506836} {"timestamp_utc": "2026-04-13T03:31:25Z", "mode": "train", "global_step": 2075, "epoch": 0.20843797086891008, "loss": 0.046, "grad_norm": 15.284062385559082, "learning_rate": 3.7151515151515156e-06, "num_tokens": 3808749.0, "completions/mean_length": 24.125, "completions/min_length": 22.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.125, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9929973483085632, "rewards/meter/std": 0.00308910827152431, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9401386976242065, "rewards/repeat_soft/std": 0.028815701603889465, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8179876804351807, "rewards/total_composite/std": 0.0048122890293598175, "reward": 0.8179876804351807, "reward_std": 0.004812279716134071, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07262216508388519, "sampling/sampling_logp_difference/max": 1.2497978210449219, "sampling/importance_sampling_ratio/min": 0.2865627110004425, "sampling/importance_sampling_ratio/mean": 1.0231256484985352, "sampling/importance_sampling_ratio/max": 1.568536639213562, "entropy": 0.6086428761482239, "clip_ratio/low_mean": 0.024852865375578403, "clip_ratio/low_min": 0.024852865375578403, "clip_ratio/high_mean": 0.01585144968703389, "clip_ratio/high_max": 0.01585144968703389, "clip_ratio/region_mean": 0.040704315062612295, "reward_total_mean": 0.8179876804351807, "reward_meter_mean": 0.9929973483085632, "reward_meter_std": 0.00308910827152431, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9401386976242065, "reward_repeat_soft_std": 0.028815701603889465, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8179876804351807, "reward_total_composite_std": 0.0048122890293598175} {"timestamp_utc": "2026-04-13T03:31:33Z", "mode": "train", "global_step": 2076, "epoch": 0.2085384229030638, "loss": 0.0349, "grad_norm": 6.402390480041504, "learning_rate": 3.7121212121212124e-06, "num_tokens": 3811253.0, "completions/mean_length": 126.0, "completions/min_length": 110.0, "completions/max_length": 146.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.0, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 146.0, "rewards/meter/mean": 0.9915544986724854, "rewards/meter/std": 0.00504684541374445, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9642132520675659, "rewards/repeat_soft/std": 0.011712804436683655, "rewards/judge_quality/mean": 0.41874998807907104, "rewards/judge_quality/std": 0.21931305527687073, "rewards/total_composite/mean": 0.8182458877563477, "rewards/total_composite/std": 0.06611452996730804, "reward": 0.8182458877563477, "reward_std": 0.06611452996730804, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09986726939678192, "sampling/sampling_logp_difference/max": 1.2332558631896973, "sampling/importance_sampling_ratio/min": 0.29134246706962585, "sampling/importance_sampling_ratio/mean": 1.0112545490264893, "sampling/importance_sampling_ratio/max": 1.9916936159133911, "entropy": 0.6253738887608051, "clip_ratio/low_mean": 0.05851300060749054, "clip_ratio/low_min": 0.05851300060749054, "clip_ratio/high_mean": 0.030942475888878107, "clip_ratio/high_max": 0.030942475888878107, "clip_ratio/region_mean": 0.08945547649636865, "reward_total_mean": 0.8182458877563477, "reward_meter_mean": 0.9915544986724854, "reward_meter_std": 0.00504684541374445, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9642132520675659, "reward_repeat_soft_std": 0.011712804436683655, "reward_judge_quality_mean": 0.41874998807907104, "reward_judge_quality_std": 0.21931305527687073, "reward_total_composite_mean": 0.8182458877563477, "reward_total_composite_std": 0.06611452996730804} {"timestamp_utc": "2026-04-13T03:31:39Z", "mode": "train", "global_step": 2077, "epoch": 0.20863887493721747, "loss": 0.0013, "grad_norm": 21.803142547607422, "learning_rate": 3.7090909090909092e-06, "num_tokens": 3812662.0, "completions/mean_length": 32.125, "completions/min_length": 28.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.125, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.7909729480743408, "rewards/meter/std": 0.28793472051620483, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9973483085632324, "rewards/repeat_soft/std": 0.0043530575931072235, "rewards/judge_quality/mean": 0.5637500286102295, "rewards/judge_quality/std": 0.22012579441070557, "rewards/total_composite/mean": 0.7747976183891296, "rewards/total_composite/std": 0.157884880900383, "reward": 0.7747976183891296, "reward_std": 0.1578848659992218, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1105106994509697, "sampling/sampling_logp_difference/max": 1.9429931640625, "sampling/importance_sampling_ratio/min": 0.1432744562625885, "sampling/importance_sampling_ratio/mean": 0.9966831803321838, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5009737312793732, "clip_ratio/low_mean": 0.02057378552854061, "clip_ratio/low_min": 0.02057378552854061, "clip_ratio/high_mean": 0.07965085003525019, "clip_ratio/high_max": 0.07965085003525019, "clip_ratio/region_mean": 0.1002246355637908, "reward_total_mean": 0.7747976183891296, "reward_meter_mean": 0.7909729480743408, "reward_meter_std": 0.28793472051620483, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9973483085632324, "reward_repeat_soft_std": 0.0043530575931072235, "reward_judge_quality_mean": 0.5637500286102295, "reward_judge_quality_std": 0.22012579441070557, "reward_total_composite_mean": 0.7747976183891296, "reward_total_composite_std": 0.157884880900383} {"timestamp_utc": "2026-04-13T03:31:47Z", "mode": "train", "global_step": 2078, "epoch": 0.20873932697137118, "loss": 0.0158, "grad_norm": 8.829496383666992, "learning_rate": 3.7060606060606065e-06, "num_tokens": 3814877.0, "completions/mean_length": 105.875, "completions/min_length": 94.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.875, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.9835255146026611, "rewards/meter/std": 0.011533156968653202, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8713139295578003, "rewards/repeat_soft/std": 0.05487740412354469, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.799342930316925, "rewards/total_composite/std": 0.020980259403586388, "reward": 0.799342930316925, "reward_std": 0.02098025195300579, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0982779785990715, "sampling/sampling_logp_difference/max": 2.2037320137023926, "sampling/importance_sampling_ratio/min": 0.11039040982723236, "sampling/importance_sampling_ratio/mean": 1.0002814531326294, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5735594071447849, "clip_ratio/low_mean": 0.03458145773038268, "clip_ratio/low_min": 0.03458145773038268, "clip_ratio/high_mean": 0.06235397141426802, "clip_ratio/high_max": 0.06235397141426802, "clip_ratio/region_mean": 0.0969354291446507, "reward_total_mean": 0.799342930316925, "reward_meter_mean": 0.9835255146026611, "reward_meter_std": 0.011533156968653202, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8713139295578003, "reward_repeat_soft_std": 0.05487740412354469, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.799342930316925, "reward_total_composite_std": 0.020980259403586388} {"timestamp_utc": "2026-04-13T03:31:58Z", "mode": "train", "global_step": 2079, "epoch": 0.20883977900552486, "loss": -0.1549, "grad_norm": 1.6853197813034058, "learning_rate": 3.7030303030303033e-06, "num_tokens": 3816480.0, "completions/mean_length": 116.375, "completions/min_length": 53.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 59.857147216796875, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.867766261100769, "rewards/meter/std": 0.3506908118724823, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9929613471031189, "rewards/repeat_soft/std": 0.009740488603711128, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.13452960550785065, "rewards/total_composite/mean": 0.7210409641265869, "rewards/total_composite/std": 0.2914007306098938, "reward": 0.7210409641265869, "reward_std": 0.2914007306098938, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12431780993938446, "sampling/sampling_logp_difference/max": 1.4401087760925293, "sampling/importance_sampling_ratio/min": 0.23690199851989746, "sampling/importance_sampling_ratio/mean": 1.0087354183197021, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6768697425723076, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.11387237254530191, "clip_ratio/high_max": 0.11387237254530191, "clip_ratio/region_mean": 0.11387237254530191, "reward_total_mean": 0.7210409641265869, "reward_meter_mean": 0.867766261100769, "reward_meter_std": 0.3506908118724823, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9929613471031189, "reward_repeat_soft_std": 0.009740488603711128, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.13452960550785065, "reward_total_composite_mean": 0.7210409641265869, "reward_total_composite_std": 0.2914007306098938} {"timestamp_utc": "2026-04-13T03:32:05Z", "mode": "train", "global_step": 2080, "epoch": 0.20894023103967854, "loss": 0.0506, "grad_norm": 7.028439044952393, "learning_rate": 3.7e-06, "num_tokens": 3819076.0, "completions/mean_length": 133.5, "completions/min_length": 113.0, "completions/max_length": 146.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 133.5, "completions/min_terminated_length": 113.0, "completions/max_terminated_length": 146.0, "rewards/meter/mean": 0.9924774169921875, "rewards/meter/std": 0.004271951038390398, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8405693173408508, "rewards/repeat_soft/std": 0.06552713364362717, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7939217686653137, "rewards/total_composite/std": 0.022882526740431786, "reward": 0.7939217686653137, "reward_std": 0.02288251742720604, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07646925747394562, "sampling/sampling_logp_difference/max": 1.9073419570922852, "sampling/importance_sampling_ratio/min": 0.14847451448440552, "sampling/importance_sampling_ratio/mean": 1.0117740631103516, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4804426319897175, "clip_ratio/low_mean": 0.013698630034923553, "clip_ratio/low_min": 0.013698630034923553, "clip_ratio/high_mean": 0.042480298317968845, "clip_ratio/high_max": 0.042480298317968845, "clip_ratio/region_mean": 0.0561789283528924, "reward_total_mean": 0.7939217686653137, "reward_meter_mean": 0.9924774169921875, "reward_meter_std": 0.004271951038390398, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8405693173408508, "reward_repeat_soft_std": 0.06552713364362717, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7939217686653137, "reward_total_composite_std": 0.022882526740431786} {"timestamp_utc": "2026-04-13T03:32:12Z", "mode": "train", "global_step": 2081, "epoch": 0.20904068307383225, "loss": -0.0045, "grad_norm": 13.313727378845215, "learning_rate": 3.6969696969696974e-06, "num_tokens": 3820668.0, "completions/mean_length": 41.0, "completions/min_length": 34.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.7364556789398193, "rewards/meter/std": 0.37101027369499207, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.977234959602356, "rewards/repeat_soft/std": 0.02546551823616028, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.7250035405158997, "rewards/total_composite/std": 0.18507851660251617, "reward": 0.7250035405158997, "reward_std": 0.18507850170135498, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13706618547439575, "sampling/sampling_logp_difference/max": 2.232736587524414, "sampling/importance_sampling_ratio/min": 0.10723457485437393, "sampling/importance_sampling_ratio/mean": 1.0035345554351807, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6771959587931633, "clip_ratio/low_mean": 0.03921568766236305, "clip_ratio/low_min": 0.03921568766236305, "clip_ratio/high_mean": 0.07592879934236407, "clip_ratio/high_max": 0.07592879934236407, "clip_ratio/region_mean": 0.11514448700472713, "reward_total_mean": 0.7250035405158997, "reward_meter_mean": 0.7364556789398193, "reward_meter_std": 0.37101027369499207, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.977234959602356, "reward_repeat_soft_std": 0.02546551823616028, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.7250035405158997, "reward_total_composite_std": 0.18507851660251617} {"timestamp_utc": "2026-04-13T03:32:18Z", "mode": "train", "global_step": 2082, "epoch": 0.20914113510798593, "loss": -0.0007, "grad_norm": 11.56971263885498, "learning_rate": 3.6939393939393942e-06, "num_tokens": 3822298.0, "completions/mean_length": 60.75, "completions/min_length": 54.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.75, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.7031545042991638, "rewards/meter/std": 0.311761736869812, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9951289892196655, "rewards/repeat_soft/std": 0.004701366648077965, "rewards/judge_quality/mean": 0.7575000524520874, "rewards/judge_quality/std": 0.1685018092393875, "rewards/total_composite/mean": 0.7931824326515198, "rewards/total_composite/std": 0.1561231017112732, "reward": 0.7931824326515198, "reward_std": 0.1561231166124344, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11900189518928528, "sampling/sampling_logp_difference/max": 2.2446751594543457, "sampling/importance_sampling_ratio/min": 0.10596195608377457, "sampling/importance_sampling_ratio/mean": 1.0141558647155762, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6608932837843895, "clip_ratio/low_mean": 0.07003264548256993, "clip_ratio/low_min": 0.07003264548256993, "clip_ratio/high_mean": 0.05079619539901614, "clip_ratio/high_max": 0.05079619539901614, "clip_ratio/region_mean": 0.12082884088158607, "reward_total_mean": 0.7931824326515198, "reward_meter_mean": 0.7031545042991638, "reward_meter_std": 0.311761736869812, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9951289892196655, "reward_repeat_soft_std": 0.004701366648077965, "reward_judge_quality_mean": 0.7575000524520874, "reward_judge_quality_std": 0.1685018092393875, "reward_total_composite_mean": 0.7931824326515198, "reward_total_composite_std": 0.1561231017112732} {"timestamp_utc": "2026-04-13T03:32:24Z", "mode": "train", "global_step": 2083, "epoch": 0.20924158714213964, "loss": 0.0084, "grad_norm": 14.409022331237793, "learning_rate": 3.690909090909091e-06, "num_tokens": 3823848.0, "completions/mean_length": 49.75, "completions/min_length": 40.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.75, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.8336257934570312, "rewards/meter/std": 0.23467960953712463, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9931490421295166, "rewards/repeat_soft/std": 0.018772710114717484, "rewards/judge_quality/mean": 0.6200000047683716, "rewards/judge_quality/std": 0.22677870094776154, "rewards/total_composite/mean": 0.8104465007781982, "rewards/total_composite/std": 0.09611405432224274, "reward": 0.8104465007781982, "reward_std": 0.09611404687166214, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1274116188287735, "sampling/sampling_logp_difference/max": 1.8613972663879395, "sampling/importance_sampling_ratio/min": 0.15545526146888733, "sampling/importance_sampling_ratio/mean": 1.0088365077972412, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6924028098583221, "clip_ratio/low_mean": 0.04416294861584902, "clip_ratio/low_min": 0.04416294861584902, "clip_ratio/high_mean": 0.06949364999309182, "clip_ratio/high_max": 0.06949364999309182, "clip_ratio/region_mean": 0.11365659860894084, "reward_total_mean": 0.8104465007781982, "reward_meter_mean": 0.8336257934570312, "reward_meter_std": 0.23467960953712463, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9931490421295166, "reward_repeat_soft_std": 0.018772710114717484, "reward_judge_quality_mean": 0.6200000047683716, "reward_judge_quality_std": 0.22677870094776154, "reward_total_composite_mean": 0.8104465007781982, "reward_total_composite_std": 0.09611405432224274} {"timestamp_utc": "2026-04-13T03:32:32Z", "mode": "train", "global_step": 2084, "epoch": 0.20934203917629332, "loss": 0.0129, "grad_norm": 7.1470489501953125, "learning_rate": 3.687878787878788e-06, "num_tokens": 3826490.0, "completions/mean_length": 138.25, "completions/min_length": 130.0, "completions/max_length": 153.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 138.25, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 153.0, "rewards/meter/mean": 0.9220691919326782, "rewards/meter/std": 0.1091160997748375, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9295242428779602, "rewards/repeat_soft/std": 0.03433368727564812, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.6688709259033203, "rewards/total_composite/std": 0.27440330386161804, "reward": 0.6688709259033203, "reward_std": 0.27440330386161804, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10410229116678238, "sampling/sampling_logp_difference/max": 2.031867027282715, "sampling/importance_sampling_ratio/min": 0.13109053671360016, "sampling/importance_sampling_ratio/mean": 1.0118486881256104, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.727693647146225, "clip_ratio/low_mean": 0.011111111380159855, "clip_ratio/low_min": 0.011111111380159855, "clip_ratio/high_mean": 0.08657465130090714, "clip_ratio/high_max": 0.08657465130090714, "clip_ratio/region_mean": 0.09768576268106699, "reward_total_mean": 0.6688709259033203, "reward_meter_mean": 0.9220691919326782, "reward_meter_std": 0.1091160997748375, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9295242428779602, "reward_repeat_soft_std": 0.03433368727564812, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.6688709259033203, "reward_total_composite_std": 0.27440330386161804} {"timestamp_utc": "2026-04-13T03:32:38Z", "mode": "train", "global_step": 2085, "epoch": 0.209442491210447, "loss": 0.0144, "grad_norm": 19.249433517456055, "learning_rate": 3.684848484848485e-06, "num_tokens": 3828073.0, "completions/mean_length": 47.875, "completions/min_length": 39.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.875, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.8721368908882141, "rewards/meter/std": 0.1489194631576538, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9862708449363708, "rewards/repeat_soft/std": 0.017856858670711517, "rewards/judge_quality/mean": 0.6599999666213989, "rewards/judge_quality/std": 0.2392846643924713, "rewards/total_composite/mean": 0.8390886783599854, "rewards/total_composite/std": 0.11134255677461624, "reward": 0.8390886783599854, "reward_std": 0.11134254932403564, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13970111310482025, "sampling/sampling_logp_difference/max": 1.8060591220855713, "sampling/importance_sampling_ratio/min": 0.1643003523349762, "sampling/importance_sampling_ratio/mean": 0.9937770366668701, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7373415306210518, "clip_ratio/low_mean": 0.047100522089749575, "clip_ratio/low_min": 0.047100522089749575, "clip_ratio/high_mean": 0.08715216629207134, "clip_ratio/high_max": 0.08715216629207134, "clip_ratio/region_mean": 0.13425268838182092, "reward_total_mean": 0.8390886783599854, "reward_meter_mean": 0.8721368908882141, "reward_meter_std": 0.1489194631576538, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9862708449363708, "reward_repeat_soft_std": 0.017856858670711517, "reward_judge_quality_mean": 0.6599999666213989, "reward_judge_quality_std": 0.2392846643924713, "reward_total_composite_mean": 0.8390886783599854, "reward_total_composite_std": 0.11134255677461624} {"timestamp_utc": "2026-04-13T03:32:45Z", "mode": "train", "global_step": 2086, "epoch": 0.2095429432446007, "loss": 0.0221, "grad_norm": 11.590227127075195, "learning_rate": 3.681818181818182e-06, "num_tokens": 3829897.0, "completions/mean_length": 57.0, "completions/min_length": 54.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.8571606278419495, "rewards/meter/std": 0.18668155372142792, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.896917998790741, "rewards/repeat_soft/std": 0.09326960891485214, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7450391054153442, "rewards/total_composite/std": 0.08667980134487152, "reward": 0.7450391054153442, "reward_std": 0.08667979389429092, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09271600097417831, "sampling/sampling_logp_difference/max": 1.3027169704437256, "sampling/importance_sampling_ratio/min": 0.27179235219955444, "sampling/importance_sampling_ratio/mean": 1.0130798816680908, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4082074984908104, "clip_ratio/low_mean": 0.02842262014746666, "clip_ratio/low_min": 0.02842262014746666, "clip_ratio/high_mean": 0.05233934987336397, "clip_ratio/high_max": 0.05233934987336397, "clip_ratio/region_mean": 0.08076197002083063, "reward_total_mean": 0.7450391054153442, "reward_meter_mean": 0.8571606278419495, "reward_meter_std": 0.18668155372142792, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.896917998790741, "reward_repeat_soft_std": 0.09326960891485214, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7450391054153442, "reward_total_composite_std": 0.08667980134487152} {"timestamp_utc": "2026-04-13T03:32:53Z", "mode": "train", "global_step": 2087, "epoch": 0.2096433952787544, "loss": 0.0645, "grad_norm": 12.176876068115234, "learning_rate": 3.678787878787879e-06, "num_tokens": 3831482.0, "completions/mean_length": 38.125, "completions/min_length": 34.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.125, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.8411445617675781, "rewards/meter/std": 0.13801231980323792, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.968995213508606, "rewards/repeat_soft/std": 0.025605278089642525, "rewards/judge_quality/mean": 0.45625001192092896, "rewards/judge_quality/std": 0.06781014055013657, "rewards/total_composite/mean": 0.7622895240783691, "rewards/total_composite/std": 0.0635455772280693, "reward": 0.7622895240783691, "reward_std": 0.0635455846786499, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12028073519468307, "sampling/sampling_logp_difference/max": 1.622389554977417, "sampling/importance_sampling_ratio/min": 0.2315794676542282, "sampling/importance_sampling_ratio/mean": 0.989653468132019, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5335894338786602, "clip_ratio/low_mean": 0.033985908376052976, "clip_ratio/low_min": 0.033985908376052976, "clip_ratio/high_mean": 0.07410045340657234, "clip_ratio/high_max": 0.07410045340657234, "clip_ratio/region_mean": 0.10808636178262532, "reward_total_mean": 0.7622895240783691, "reward_meter_mean": 0.8411445617675781, "reward_meter_std": 0.13801231980323792, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.968995213508606, "reward_repeat_soft_std": 0.025605278089642525, "reward_judge_quality_mean": 0.45625001192092896, "reward_judge_quality_std": 0.06781014055013657, "reward_total_composite_mean": 0.7622895240783691, "reward_total_composite_std": 0.0635455772280693} {"timestamp_utc": "2026-04-13T03:32:59Z", "mode": "train", "global_step": 2088, "epoch": 0.2097438473129081, "loss": -0.0186, "grad_norm": 5.901555061340332, "learning_rate": 3.6757575757575757e-06, "num_tokens": 3833181.0, "completions/mean_length": 56.375, "completions/min_length": 51.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.375, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9172972440719604, "rewards/meter/std": 0.11501458287239075, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.959983229637146, "rewards/repeat_soft/std": 0.021089302375912666, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.8057820796966553, "rewards/total_composite/std": 0.08057688921689987, "reward": 0.8057820796966553, "reward_std": 0.08057688921689987, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08631671220064163, "sampling/sampling_logp_difference/max": 1.3298513889312744, "sampling/importance_sampling_ratio/min": 0.26451656222343445, "sampling/importance_sampling_ratio/mean": 1.0073020458221436, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.441681444644928, "clip_ratio/low_mean": 0.040508512407541275, "clip_ratio/low_min": 0.040508512407541275, "clip_ratio/high_mean": 0.062497069127857685, "clip_ratio/high_max": 0.062497069127857685, "clip_ratio/region_mean": 0.10300558153539896, "reward_total_mean": 0.8057820796966553, "reward_meter_mean": 0.9172972440719604, "reward_meter_std": 0.11501458287239075, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.959983229637146, "reward_repeat_soft_std": 0.021089302375912666, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.8057820796966553, "reward_total_composite_std": 0.08057688921689987} {"timestamp_utc": "2026-04-13T03:33:07Z", "mode": "train", "global_step": 2089, "epoch": 0.20984429934706178, "loss": 0.0459, "grad_norm": 10.162285804748535, "learning_rate": 3.672727272727273e-06, "num_tokens": 3835114.0, "completions/mean_length": 78.625, "completions/min_length": 67.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.625, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.9665009379386902, "rewards/meter/std": 0.022097352892160416, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9312648773193359, "rewards/repeat_soft/std": 0.03170910105109215, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8051769733428955, "rewards/total_composite/std": 0.010977999307215214, "reward": 0.8051769733428955, "reward_std": 0.010978000238537788, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.126860573887825, "sampling/sampling_logp_difference/max": 1.4295289516448975, "sampling/importance_sampling_ratio/min": 0.23942168056964874, "sampling/importance_sampling_ratio/mean": 1.0192826986312866, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7244798317551613, "clip_ratio/low_mean": 0.055081434547901154, "clip_ratio/low_min": 0.055081434547901154, "clip_ratio/high_mean": 0.09016945492476225, "clip_ratio/high_max": 0.09016945492476225, "clip_ratio/region_mean": 0.1452508894726634, "reward_total_mean": 0.8051769733428955, "reward_meter_mean": 0.9665009379386902, "reward_meter_std": 0.022097352892160416, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9312648773193359, "reward_repeat_soft_std": 0.03170910105109215, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8051769733428955, "reward_total_composite_std": 0.010977999307215214} {"timestamp_utc": "2026-04-13T03:33:25Z", "mode": "train", "global_step": 2090, "epoch": 0.20994475138121546, "loss": -0.1949, "grad_norm": 2.1552467346191406, "learning_rate": 3.6696969696969697e-06, "num_tokens": 3837186.0, "completions/mean_length": 151.0, "completions/min_length": 89.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 99.42857360839844, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.8501554727554321, "rewards/meter/std": 0.27195051312446594, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9265455007553101, "rewards/repeat_soft/std": 0.03641831502318382, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.693107008934021, "rewards/total_composite/std": 0.28227537870407104, "reward": 0.693107008934021, "reward_std": 0.28227537870407104, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1279471069574356, "sampling/sampling_logp_difference/max": 1.6512885093688965, "sampling/importance_sampling_ratio/min": 0.19180260598659515, "sampling/importance_sampling_ratio/mean": 1.0164985656738281, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5505548864603043, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10530572757124901, "clip_ratio/high_max": 0.10530572757124901, "clip_ratio/region_mean": 0.10530572757124901, "reward_total_mean": 0.693107008934021, "reward_meter_mean": 0.8501554727554321, "reward_meter_std": 0.27195051312446594, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.2651650309562683, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9265455007553101, "reward_repeat_soft_std": 0.03641831502318382, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.693107008934021, "reward_total_composite_std": 0.28227537870407104} {"timestamp_utc": "2026-04-13T03:33:36Z", "mode": "train", "global_step": 2091, "epoch": 0.21004520341536917, "loss": -0.1028, "grad_norm": 2.1555421352386475, "learning_rate": 3.6666666666666666e-06, "num_tokens": 3838638.0, "completions/mean_length": 163.5, "completions/min_length": 42.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 47.333335876464844, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.6819493770599365, "rewards/meter/std": 0.3747018873691559, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9909876585006714, "rewards/repeat_soft/std": 0.01207340694963932, "rewards/judge_quality/mean": 0.3174999952316284, "rewards/judge_quality/std": 0.17782413959503174, "rewards/total_composite/mean": 0.4877339005470276, "rewards/total_composite/std": 0.33767566084861755, "reward": 0.4877339005470276, "reward_std": 0.33767563104629517, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14304500818252563, "sampling/sampling_logp_difference/max": 2.136958122253418, "sampling/importance_sampling_ratio/min": 0.11801327764987946, "sampling/importance_sampling_ratio/mean": 1.0031154155731201, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6254238933324814, "clip_ratio/low_mean": 0.030632411129772663, "clip_ratio/low_min": 0.030632411129772663, "clip_ratio/high_mean": 0.07301619462668896, "clip_ratio/high_max": 0.07301619462668896, "clip_ratio/region_mean": 0.10364860575646162, "reward_total_mean": 0.4877339005470276, "reward_meter_mean": 0.6819493770599365, "reward_meter_std": 0.3747018873691559, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9909876585006714, "reward_repeat_soft_std": 0.01207340694963932, "reward_judge_quality_mean": 0.3174999952316284, "reward_judge_quality_std": 0.17782413959503174, "reward_total_composite_mean": 0.4877339005470276, "reward_total_composite_std": 0.33767566084861755} {"timestamp_utc": "2026-04-13T03:33:42Z", "mode": "train", "global_step": 2092, "epoch": 0.21014565544952285, "loss": 0.0111, "grad_norm": 13.458623886108398, "learning_rate": 3.6636363636363643e-06, "num_tokens": 3840205.0, "completions/mean_length": 48.875, "completions/min_length": 45.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.875, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9532315731048584, "rewards/meter/std": 0.05562817305326462, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9905248880386353, "rewards/repeat_soft/std": 0.009151238016784191, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.10392305999994278, "rewards/total_composite/mean": 0.8175066709518433, "rewards/total_composite/std": 0.0443679578602314, "reward": 0.8175066709518433, "reward_std": 0.044367946684360504, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12491103261709213, "sampling/sampling_logp_difference/max": 1.3542770147323608, "sampling/importance_sampling_ratio/min": 0.2581338584423065, "sampling/importance_sampling_ratio/mean": 0.9829036593437195, "sampling/importance_sampling_ratio/max": 1.9256552457809448, "entropy": 0.7067362740635872, "clip_ratio/low_mean": 0.026492317439988256, "clip_ratio/low_min": 0.026492317439988256, "clip_ratio/high_mean": 0.0974433058872819, "clip_ratio/high_max": 0.0974433058872819, "clip_ratio/region_mean": 0.12393562332727015, "reward_total_mean": 0.8175066709518433, "reward_meter_mean": 0.9532315731048584, "reward_meter_std": 0.05562817305326462, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9905248880386353, "reward_repeat_soft_std": 0.009151238016784191, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.10392305999994278, "reward_total_composite_mean": 0.8175066709518433, "reward_total_composite_std": 0.0443679578602314} {"timestamp_utc": "2026-04-13T03:33:50Z", "mode": "train", "global_step": 2093, "epoch": 0.21024610748367653, "loss": 0.0379, "grad_norm": 7.430802822113037, "learning_rate": 3.660606060606061e-06, "num_tokens": 3842684.0, "completions/mean_length": 120.875, "completions/min_length": 104.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.875, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.7832175493240356, "rewards/meter/std": 0.3000604808330536, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9654523730278015, "rewards/repeat_soft/std": 0.022229457274079323, "rewards/judge_quality/mean": 0.4699999690055847, "rewards/judge_quality/std": 0.1414213627576828, "rewards/total_composite/mean": 0.7353056073188782, "rewards/total_composite/std": 0.15962445735931396, "reward": 0.7353056073188782, "reward_std": 0.15962445735931396, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10332570970058441, "sampling/sampling_logp_difference/max": 1.7874021530151367, "sampling/importance_sampling_ratio/min": 0.1673944741487503, "sampling/importance_sampling_ratio/mean": 0.9941654801368713, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6367462798953056, "clip_ratio/low_mean": 0.0225705336779356, "clip_ratio/low_min": 0.0225705336779356, "clip_ratio/high_mean": 0.08684237208217382, "clip_ratio/high_max": 0.08684237208217382, "clip_ratio/region_mean": 0.10941290576010942, "reward_total_mean": 0.7353056073188782, "reward_meter_mean": 0.7832175493240356, "reward_meter_std": 0.3000604808330536, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9654523730278015, "reward_repeat_soft_std": 0.022229457274079323, "reward_judge_quality_mean": 0.4699999690055847, "reward_judge_quality_std": 0.1414213627576828, "reward_total_composite_mean": 0.7353056073188782, "reward_total_composite_std": 0.15962445735931396} {"timestamp_utc": "2026-04-13T03:33:56Z", "mode": "train", "global_step": 2094, "epoch": 0.21034655951783024, "loss": -0.0223, "grad_norm": 13.651933670043945, "learning_rate": 3.657575757575758e-06, "num_tokens": 3844134.0, "completions/mean_length": 29.25, "completions/min_length": 26.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.25, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.8039599657058716, "rewards/meter/std": 0.2616676092147827, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9610613584518433, "rewards/repeat_soft/std": 0.004068940877914429, "rewards/judge_quality/mean": 0.375, "rewards/judge_quality/std": 0.1035098284482956, "rewards/total_composite/mean": 0.7203880548477173, "rewards/total_composite/std": 0.1031971275806427, "reward": 0.7203880548477173, "reward_std": 0.1031971201300621, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10702598094940186, "sampling/sampling_logp_difference/max": 0.6543436050415039, "sampling/importance_sampling_ratio/min": 0.5197831392288208, "sampling/importance_sampling_ratio/mean": 1.0146981477737427, "sampling/importance_sampling_ratio/max": 1.8462437391281128, "entropy": 0.8062855005264282, "clip_ratio/low_mean": 0.04040873982012272, "clip_ratio/low_min": 0.04040873982012272, "clip_ratio/high_mean": 0.05421301443129778, "clip_ratio/high_max": 0.05421301443129778, "clip_ratio/region_mean": 0.0946217542514205, "reward_total_mean": 0.7203880548477173, "reward_meter_mean": 0.8039599657058716, "reward_meter_std": 0.2616676092147827, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9610613584518433, "reward_repeat_soft_std": 0.004068940877914429, "reward_judge_quality_mean": 0.375, "reward_judge_quality_std": 0.1035098284482956, "reward_total_composite_mean": 0.7203880548477173, "reward_total_composite_std": 0.1031971275806427} {"timestamp_utc": "2026-04-13T03:34:03Z", "mode": "train", "global_step": 2095, "epoch": 0.21044701155198392, "loss": 0.1289, "grad_norm": 8.956899642944336, "learning_rate": 3.654545454545455e-06, "num_tokens": 3846101.0, "completions/mean_length": 82.875, "completions/min_length": 67.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.875, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.9618838429450989, "rewards/meter/std": 0.028195206075906754, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8653395175933838, "rewards/repeat_soft/std": 0.09957475960254669, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7827566862106323, "rewards/total_composite/std": 0.03725456818938255, "reward": 0.7827566862106323, "reward_std": 0.037254586815834045, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08464358001947403, "sampling/sampling_logp_difference/max": 1.505594253540039, "sampling/importance_sampling_ratio/min": 0.22188539803028107, "sampling/importance_sampling_ratio/mean": 1.0049885511398315, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5453535169363022, "clip_ratio/low_mean": 0.0222179489210248, "clip_ratio/low_min": 0.0222179489210248, "clip_ratio/high_mean": 0.07322503253817558, "clip_ratio/high_max": 0.07322503253817558, "clip_ratio/region_mean": 0.09544298145920038, "reward_total_mean": 0.7827566862106323, "reward_meter_mean": 0.9618838429450989, "reward_meter_std": 0.028195206075906754, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8653395175933838, "reward_repeat_soft_std": 0.09957475960254669, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7827566862106323, "reward_total_composite_std": 0.03725456818938255} {"timestamp_utc": "2026-04-13T03:34:10Z", "mode": "train", "global_step": 2096, "epoch": 0.21054746358613763, "loss": -0.0369, "grad_norm": 19.034269332885742, "learning_rate": 3.651515151515152e-06, "num_tokens": 3847615.0, "completions/mean_length": 32.25, "completions/min_length": 24.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.25, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.8868011236190796, "rewards/meter/std": 0.27957969903945923, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9538223147392273, "rewards/repeat_soft/std": 0.01614215224981308, "rewards/judge_quality/mean": 0.7987500429153442, "rewards/judge_quality/std": 0.22465452551841736, "rewards/total_composite/mean": 0.8840677738189697, "rewards/total_composite/std": 0.18048609793186188, "reward": 0.8840677738189697, "reward_std": 0.18048609793186188, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14622631669044495, "sampling/sampling_logp_difference/max": 1.1833381652832031, "sampling/importance_sampling_ratio/min": 0.30625471472740173, "sampling/importance_sampling_ratio/mean": 1.0226842164993286, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0431586354970932, "clip_ratio/low_mean": 0.038541666232049465, "clip_ratio/low_min": 0.038541666232049465, "clip_ratio/high_mean": 0.13522013509646058, "clip_ratio/high_max": 0.13522013509646058, "clip_ratio/region_mean": 0.17376180132851005, "reward_total_mean": 0.8840677738189697, "reward_meter_mean": 0.8868011236190796, "reward_meter_std": 0.27957969903945923, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9538223147392273, "reward_repeat_soft_std": 0.01614215224981308, "reward_judge_quality_mean": 0.7987500429153442, "reward_judge_quality_std": 0.22465452551841736, "reward_total_composite_mean": 0.8840677738189697, "reward_total_composite_std": 0.18048609793186188} {"timestamp_utc": "2026-04-13T03:34:16Z", "mode": "train", "global_step": 2097, "epoch": 0.2106479156202913, "loss": -0.0506, "grad_norm": 10.256195068359375, "learning_rate": 3.648484848484849e-06, "num_tokens": 3849175.0, "completions/mean_length": 50.0, "completions/min_length": 40.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.0, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9866857528686523, "rewards/meter/std": 0.011292714625597, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9944052696228027, "rewards/repeat_soft/std": 0.004660876002162695, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.1011011004447937, "rewards/total_composite/mean": 0.8351991176605225, "rewards/total_composite/std": 0.032422903925180435, "reward": 0.8351991176605225, "reward_std": 0.03242290019989014, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10459836572408676, "sampling/sampling_logp_difference/max": 1.197396993637085, "sampling/importance_sampling_ratio/min": 0.3019792437553406, "sampling/importance_sampling_ratio/mean": 0.9995690584182739, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5753679424524307, "clip_ratio/low_mean": 0.07244687154889107, "clip_ratio/low_min": 0.07244687154889107, "clip_ratio/high_mean": 0.012500000186264515, "clip_ratio/high_max": 0.012500000186264515, "clip_ratio/region_mean": 0.08494687173515558, "reward_total_mean": 0.8351991176605225, "reward_meter_mean": 0.9866857528686523, "reward_meter_std": 0.011292714625597, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9944052696228027, "reward_repeat_soft_std": 0.004660876002162695, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.1011011004447937, "reward_total_composite_mean": 0.8351991176605225, "reward_total_composite_std": 0.032422903925180435} {"timestamp_utc": "2026-04-13T03:34:28Z", "mode": "train", "global_step": 2098, "epoch": 0.210748367654445, "loss": -0.1456, "grad_norm": 1.7596557140350342, "learning_rate": 3.645454545454546e-06, "num_tokens": 3850889.0, "completions/mean_length": 114.25, "completions/min_length": 48.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 57.42857360839844, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9097956418991089, "rewards/meter/std": 0.24013350903987885, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9737592935562134, "rewards/repeat_soft/std": 0.027391759678721428, "rewards/judge_quality/mean": 0.6899999976158142, "rewards/judge_quality/std": 0.33903223276138306, "rewards/total_composite/mean": 0.812912106513977, "rewards/total_composite/std": 0.3351987302303314, "reward": 0.812912106513977, "reward_std": 0.33519870042800903, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10081593692302704, "sampling/sampling_logp_difference/max": 1.1449148654937744, "sampling/importance_sampling_ratio/min": 0.31825101375579834, "sampling/importance_sampling_ratio/mean": 1.0137300491333008, "sampling/importance_sampling_ratio/max": 1.9089596271514893, "entropy": 0.565786324441433, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09106788039207458, "clip_ratio/high_max": 0.09106788039207458, "clip_ratio/region_mean": 0.09106788039207458, "reward_total_mean": 0.812912106513977, "reward_meter_mean": 0.9097956418991089, "reward_meter_std": 0.24013350903987885, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9737592935562134, "reward_repeat_soft_std": 0.027391759678721428, "reward_judge_quality_mean": 0.6899999976158142, "reward_judge_quality_std": 0.33903223276138306, "reward_total_composite_mean": 0.812912106513977, "reward_total_composite_std": 0.3351987302303314} {"timestamp_utc": "2026-04-13T03:34:39Z", "mode": "train", "global_step": 2099, "epoch": 0.2108488196885987, "loss": -0.0945, "grad_norm": 2.0721750259399414, "learning_rate": 3.642424242424243e-06, "num_tokens": 3852419.0, "completions/mean_length": 98.25, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 39.142860412597656, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.5845903158187866, "rewards/meter/std": 0.3937879502773285, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9654804468154907, "rewards/repeat_soft/std": 0.03213111683726311, "rewards/judge_quality/mean": 0.44749999046325684, "rewards/judge_quality/std": 0.19840435683727264, "rewards/total_composite/mean": 0.6107386946678162, "rewards/total_composite/std": 0.2815038561820984, "reward": 0.6107386946678162, "reward_std": 0.2815038561820984, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09922963380813599, "sampling/sampling_logp_difference/max": 2.3520638942718506, "sampling/importance_sampling_ratio/min": 0.09517253935337067, "sampling/importance_sampling_ratio/mean": 1.0107407569885254, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4105829447507858, "clip_ratio/low_mean": 0.018002322874963284, "clip_ratio/low_min": 0.018002322874963284, "clip_ratio/high_mean": 0.06298434268683195, "clip_ratio/high_max": 0.06298434268683195, "clip_ratio/region_mean": 0.08098666556179523, "reward_total_mean": 0.6107386946678162, "reward_meter_mean": 0.5845903158187866, "reward_meter_std": 0.3937879502773285, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9654804468154907, "reward_repeat_soft_std": 0.03213111683726311, "reward_judge_quality_mean": 0.44749999046325684, "reward_judge_quality_std": 0.19840435683727264, "reward_total_composite_mean": 0.6107386946678162, "reward_total_composite_std": 0.2815038561820984} {"timestamp_utc": "2026-04-13T03:34:47Z", "mode": "train", "global_step": 2100, "epoch": 0.21094927172275238, "loss": 0.1089, "grad_norm": 7.32354211807251, "learning_rate": 3.6393939393939398e-06, "num_tokens": 3854842.0, "completions/mean_length": 129.875, "completions/min_length": 114.0, "completions/max_length": 157.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 129.875, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.9839904308319092, "rewards/meter/std": 0.005569413769990206, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9767938256263733, "rewards/repeat_soft/std": 0.015302787534892559, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.8296000957489014, "rewards/total_composite/std": 0.050149329006671906, "reward": 0.8296000957489014, "reward_std": 0.05014932155609131, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11597942560911179, "sampling/sampling_logp_difference/max": 1.8400368690490723, "sampling/importance_sampling_ratio/min": 0.1588115692138672, "sampling/importance_sampling_ratio/mean": 1.0166877508163452, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7351525127887726, "clip_ratio/low_mean": 0.0795178646221757, "clip_ratio/low_min": 0.0795178646221757, "clip_ratio/high_mean": 0.01864035055041313, "clip_ratio/high_max": 0.01864035055041313, "clip_ratio/region_mean": 0.09815821517258883, "reward_total_mean": 0.8296000957489014, "reward_meter_mean": 0.9839904308319092, "reward_meter_std": 0.005569413769990206, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9767938256263733, "reward_repeat_soft_std": 0.015302787534892559, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.8296000957489014, "reward_total_composite_std": 0.050149329006671906} {"timestamp_utc": "2026-04-13T03:35:38Z", "mode": "eval", "global_step": 2100, "epoch": 0.21094927172275238, "eval_loss": NaN, "eval_runtime": 50.8834, "eval_samples_per_second": 1.572, "eval_steps_per_second": 0.197, "eval_num_tokens": 3854842.0, "eval_completions/mean_length": 97.2, "eval_completions/min_length": 40.0, "eval_completions/max_length": 194.5, "eval_completions/clipped_ratio": 0.0125, "eval_completions/mean_terminated_length": 92.01428604125977, "eval_completions/min_terminated_length": 40.0, "eval_completions/max_terminated_length": 160.8, "eval_rewards/meter/mean": 0.7677051365375519, "eval_rewards/meter/std": 0.3037183128297329, "eval_rewards/count_adherence/mean": 0.9410416722297669, "eval_rewards/count_adherence/std": 0.13017416037619114, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.07071067690849304, "eval_rewards/repeat_soft/mean": 0.961785352230072, "eval_rewards/repeat_soft/std": 0.03912730384618044, "eval_rewards/judge_quality/mean": 0.44649999439716337, "eval_rewards/judge_quality/std": 0.14057933762669564, "eval_rewards/total_composite/mean": 0.7057932257652283, "eval_rewards/total_composite/std": 0.1651357538998127, "eval_reward": 0.7057932257652283, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.06115949191153049, "eval_sampling/sampling_logp_difference/max": 1.1165693283081055, "eval_sampling/importance_sampling_ratio/min": 0.33423389196395875, "eval_sampling/importance_sampling_ratio/mean": 1.014259397983551, "eval_sampling/importance_sampling_ratio/max": 1.4493134021759033, "eval_entropy": 0.6802726030349732, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7057932257652283, "eval_reward_meter_mean": 0.7677051365375519, "eval_reward_meter_std": 0.3037183128297329, "eval_reward_count_adherence_mean": 0.9410416722297669, "eval_reward_count_adherence_std": 0.13017416037619114, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.07071067690849304, "eval_reward_repeat_soft_mean": 0.961785352230072, "eval_reward_repeat_soft_std": 0.03912730384618044, "eval_reward_judge_quality_mean": 0.44649999439716337, "eval_reward_judge_quality_std": 0.14057933762669564, "eval_reward_total_composite_mean": 0.7057932257652283, "eval_reward_total_composite_std": 0.1651357538998127} {"timestamp_utc": "2026-04-13T03:35:53Z", "mode": "train", "global_step": 2101, "epoch": 0.2110497237569061, "loss": 0.0385, "grad_norm": 9.28287410736084, "learning_rate": 3.6363636363636366e-06, "num_tokens": 3856697.0, "completions/mean_length": 75.875, "completions/min_length": 68.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.875, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9026280641555786, "rewards/meter/std": 0.22750382125377655, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9775274991989136, "rewards/repeat_soft/std": 0.01958920620381832, "rewards/judge_quality/mean": 0.5362499952316284, "rewards/judge_quality/std": 0.22385822236537933, "rewards/total_composite/mean": 0.8148103952407837, "rewards/total_composite/std": 0.14638440310955048, "reward": 0.8148103952407837, "reward_std": 0.14638440310955048, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12181781977415085, "sampling/sampling_logp_difference/max": 1.99855375289917, "sampling/importance_sampling_ratio/min": 0.13553115725517273, "sampling/importance_sampling_ratio/mean": 1.0055389404296875, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8098498359322548, "clip_ratio/low_mean": 0.023878044448792934, "clip_ratio/low_min": 0.023878044448792934, "clip_ratio/high_mean": 0.07689445745199919, "clip_ratio/high_max": 0.07689445745199919, "clip_ratio/region_mean": 0.10077250190079212, "reward_total_mean": 0.8148103952407837, "reward_meter_mean": 0.9026280641555786, "reward_meter_std": 0.22750382125377655, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9775274991989136, "reward_repeat_soft_std": 0.01958920620381832, "reward_judge_quality_mean": 0.5362499952316284, "reward_judge_quality_std": 0.22385822236537933, "reward_total_composite_mean": 0.8148103952407837, "reward_total_composite_std": 0.14638440310955048} {"timestamp_utc": "2026-04-13T03:36:00Z", "mode": "train", "global_step": 2102, "epoch": 0.21115017579105977, "loss": 0.1691, "grad_norm": 12.889764785766602, "learning_rate": 3.633333333333334e-06, "num_tokens": 3858608.0, "completions/mean_length": 62.875, "completions/min_length": 49.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.875, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.6229853630065918, "rewards/meter/std": 0.3268055021762848, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9647608995437622, "rewards/repeat_soft/std": 0.030978718772530556, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.6738194823265076, "rewards/total_composite/std": 0.15876537561416626, "reward": 0.6738194823265076, "reward_std": 0.15876537561416626, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14864981174468994, "sampling/sampling_logp_difference/max": 1.5162372589111328, "sampling/importance_sampling_ratio/min": 0.2195364087820053, "sampling/importance_sampling_ratio/mean": 1.0199612379074097, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9885671213269234, "clip_ratio/low_mean": 0.030481284484267235, "clip_ratio/low_min": 0.030481284484267235, "clip_ratio/high_mean": 0.1150286365300417, "clip_ratio/high_max": 0.1150286365300417, "clip_ratio/region_mean": 0.14550992101430893, "reward_total_mean": 0.6738194823265076, "reward_meter_mean": 0.6229853630065918, "reward_meter_std": 0.3268055021762848, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9647608995437622, "reward_repeat_soft_std": 0.030978718772530556, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.6738194823265076, "reward_total_composite_std": 0.15876537561416626} {"timestamp_utc": "2026-04-13T03:36:06Z", "mode": "train", "global_step": 2103, "epoch": 0.21125062782521345, "loss": 0.052, "grad_norm": 7.861645221710205, "learning_rate": 3.6303030303030307e-06, "num_tokens": 3860378.0, "completions/mean_length": 57.25, "completions/min_length": 48.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.25, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9749475121498108, "rewards/meter/std": 0.03681611269712448, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9833121299743652, "rewards/repeat_soft/std": 0.026857227087020874, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.8329325914382935, "rewards/total_composite/std": 0.05855702981352806, "reward": 0.8329325914382935, "reward_std": 0.058557022362947464, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08695106208324432, "sampling/sampling_logp_difference/max": 1.3982677459716797, "sampling/importance_sampling_ratio/min": 0.2470245212316513, "sampling/importance_sampling_ratio/mean": 1.0184166431427002, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6275135800242424, "clip_ratio/low_mean": 0.06841344479471445, "clip_ratio/low_min": 0.06841344479471445, "clip_ratio/high_mean": 0.009615384973585606, "clip_ratio/high_max": 0.009615384973585606, "clip_ratio/region_mean": 0.07802882976830006, "reward_total_mean": 0.8329325914382935, "reward_meter_mean": 0.9749475121498108, "reward_meter_std": 0.03681611269712448, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9833121299743652, "reward_repeat_soft_std": 0.026857227087020874, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.8329325914382935, "reward_total_composite_std": 0.05855702981352806} {"timestamp_utc": "2026-04-13T03:36:20Z", "mode": "train", "global_step": 2104, "epoch": 0.21135107985936716, "loss": -0.1624, "grad_norm": 3.4953670501708984, "learning_rate": 3.6272727272727275e-06, "num_tokens": 3862452.0, "completions/mean_length": 145.25, "completions/min_length": 79.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 92.85714721679688, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.36155641078948975, "rewards/meter/std": 0.30475810170173645, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.25877460837364197, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9763386249542236, "rewards/repeat_soft/std": 0.01579638011753559, "rewards/judge_quality/mean": 0.44875001907348633, "rewards/judge_quality/std": 0.2105392962694168, "rewards/total_composite/mean": 0.4981597065925598, "rewards/total_composite/std": 0.2359398752450943, "reward": 0.4981597065925598, "reward_std": 0.2359398603439331, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1555565446615219, "sampling/sampling_logp_difference/max": 3.1908135414123535, "sampling/importance_sampling_ratio/min": 0.04113839194178581, "sampling/importance_sampling_ratio/mean": 0.9900609254837036, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5108491368591785, "clip_ratio/low_mean": 0.055379404220730066, "clip_ratio/low_min": 0.055379404220730066, "clip_ratio/high_mean": 0.07649858482182026, "clip_ratio/high_max": 0.07649858482182026, "clip_ratio/region_mean": 0.13187798904255033, "reward_total_mean": 0.4981597065925598, "reward_meter_mean": 0.36155641078948975, "reward_meter_std": 0.30475810170173645, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.25877460837364197, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9763386249542236, "reward_repeat_soft_std": 0.01579638011753559, "reward_judge_quality_mean": 0.44875001907348633, "reward_judge_quality_std": 0.2105392962694168, "reward_total_composite_mean": 0.4981597065925598, "reward_total_composite_std": 0.2359398752450943} {"timestamp_utc": "2026-04-13T03:36:27Z", "mode": "train", "global_step": 2105, "epoch": 0.21145153189352084, "loss": 0.0853, "grad_norm": 17.789051055908203, "learning_rate": 3.6242424242424248e-06, "num_tokens": 3864232.0, "completions/mean_length": 55.5, "completions/min_length": 44.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.5, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.8470145463943481, "rewards/meter/std": 0.3288153111934662, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9872956275939941, "rewards/repeat_soft/std": 0.02068096026778221, "rewards/judge_quality/mean": 0.5099999904632568, "rewards/judge_quality/std": 0.1302744597196579, "rewards/total_composite/mean": 0.7828861474990845, "rewards/total_composite/std": 0.1617753952741623, "reward": 0.7828861474990845, "reward_std": 0.1617753952741623, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15126614272594452, "sampling/sampling_logp_difference/max": 2.9295201301574707, "sampling/importance_sampling_ratio/min": 0.05342266708612442, "sampling/importance_sampling_ratio/mean": 1.0185637474060059, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7761992886662483, "clip_ratio/low_mean": 0.027721773833036423, "clip_ratio/low_min": 0.027721773833036423, "clip_ratio/high_mean": 0.10675699450075626, "clip_ratio/high_max": 0.10675699450075626, "clip_ratio/region_mean": 0.1344787683337927, "reward_total_mean": 0.7828861474990845, "reward_meter_mean": 0.8470145463943481, "reward_meter_std": 0.3288153111934662, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9872956275939941, "reward_repeat_soft_std": 0.02068096026778221, "reward_judge_quality_mean": 0.5099999904632568, "reward_judge_quality_std": 0.1302744597196579, "reward_total_composite_mean": 0.7828861474990845, "reward_total_composite_std": 0.1617753952741623} {"timestamp_utc": "2026-04-13T03:36:34Z", "mode": "train", "global_step": 2106, "epoch": 0.21155198392767455, "loss": -0.017, "grad_norm": 9.659167289733887, "learning_rate": 3.6212121212121216e-06, "num_tokens": 3866231.0, "completions/mean_length": 82.875, "completions/min_length": 73.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.875, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.766258716583252, "rewards/meter/std": 0.3792426586151123, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9681954383850098, "rewards/repeat_soft/std": 0.01042378880083561, "rewards/judge_quality/mean": 0.6074999570846558, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.7738860249519348, "rewards/total_composite/std": 0.17947492003440857, "reward": 0.7738860249519348, "reward_std": 0.17947492003440857, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12856519222259521, "sampling/sampling_logp_difference/max": 3.3004555702209473, "sampling/importance_sampling_ratio/min": 0.03686637058854103, "sampling/importance_sampling_ratio/mean": 1.0160131454467773, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7579992637038231, "clip_ratio/low_mean": 0.021439709700644016, "clip_ratio/low_min": 0.021439709700644016, "clip_ratio/high_mean": 0.0807711984962225, "clip_ratio/high_max": 0.0807711984962225, "clip_ratio/region_mean": 0.10221090819686651, "reward_total_mean": 0.7738860249519348, "reward_meter_mean": 0.766258716583252, "reward_meter_std": 0.3792426586151123, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9681954383850098, "reward_repeat_soft_std": 0.01042378880083561, "reward_judge_quality_mean": 0.6074999570846558, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.7738860249519348, "reward_total_composite_std": 0.17947492003440857} {"timestamp_utc": "2026-04-13T03:36:41Z", "mode": "train", "global_step": 2107, "epoch": 0.21165243596182823, "loss": -0.0028, "grad_norm": 18.998262405395508, "learning_rate": 3.6181818181818184e-06, "num_tokens": 3867915.0, "completions/mean_length": 45.5, "completions/min_length": 41.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.5, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9616931676864624, "rewards/meter/std": 0.06082873418927193, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9575510621070862, "rewards/repeat_soft/std": 0.03221066668629646, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8112670183181763, "rewards/total_composite/std": 0.025647437199950218, "reward": 0.8112670183181763, "reward_std": 0.02564743347465992, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11388992518186569, "sampling/sampling_logp_difference/max": 1.8165251016616821, "sampling/importance_sampling_ratio/min": 0.16258975863456726, "sampling/importance_sampling_ratio/mean": 0.9958945512771606, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6555005386471748, "clip_ratio/low_mean": 0.013707729522138834, "clip_ratio/low_min": 0.013707729522138834, "clip_ratio/high_mean": 0.10087727475911379, "clip_ratio/high_max": 0.10087727475911379, "clip_ratio/region_mean": 0.11458500428125262, "reward_total_mean": 0.8112670183181763, "reward_meter_mean": 0.9616931676864624, "reward_meter_std": 0.06082873418927193, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9575510621070862, "reward_repeat_soft_std": 0.03221066668629646, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8112670183181763, "reward_total_composite_std": 0.025647437199950218} {"timestamp_utc": "2026-04-13T03:36:47Z", "mode": "train", "global_step": 2108, "epoch": 0.2117528879959819, "loss": -0.0634, "grad_norm": 11.53836441040039, "learning_rate": 3.6151515151515153e-06, "num_tokens": 3869712.0, "completions/mean_length": 51.625, "completions/min_length": 41.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.625, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.7280347347259521, "rewards/meter/std": 0.3798750638961792, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9862743616104126, "rewards/repeat_soft/std": 0.011346034705638885, "rewards/judge_quality/mean": 0.5737500190734863, "rewards/judge_quality/std": 0.1566559225320816, "rewards/total_composite/mean": 0.7483680248260498, "rewards/total_composite/std": 0.1807517111301422, "reward": 0.7483680248260498, "reward_std": 0.18075168132781982, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.135197252035141, "sampling/sampling_logp_difference/max": 3.7162930965423584, "sampling/importance_sampling_ratio/min": 0.024323968216776848, "sampling/importance_sampling_ratio/mean": 1.0072294473648071, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6976098716259003, "clip_ratio/low_mean": 0.05076433997601271, "clip_ratio/low_min": 0.05076433997601271, "clip_ratio/high_mean": 0.05098658613860607, "clip_ratio/high_max": 0.05098658613860607, "clip_ratio/region_mean": 0.10175092611461878, "reward_total_mean": 0.7483680248260498, "reward_meter_mean": 0.7280347347259521, "reward_meter_std": 0.3798750638961792, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9862743616104126, "reward_repeat_soft_std": 0.011346034705638885, "reward_judge_quality_mean": 0.5737500190734863, "reward_judge_quality_std": 0.1566559225320816, "reward_total_composite_mean": 0.7483680248260498, "reward_total_composite_std": 0.1807517111301422} {"timestamp_utc": "2026-04-13T03:36:54Z", "mode": "train", "global_step": 2109, "epoch": 0.21185334003013562, "loss": 0.0734, "grad_norm": 20.51703643798828, "learning_rate": 3.6121212121212125e-06, "num_tokens": 3871250.0, "completions/mean_length": 36.25, "completions/min_length": 35.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.25, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.7420197129249573, "rewards/meter/std": 0.33649101853370667, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9965255260467529, "rewards/repeat_soft/std": 0.007875239476561546, "rewards/judge_quality/mean": 0.877500057220459, "rewards/judge_quality/std": 0.07869471609592438, "rewards/total_composite/mean": 0.8468114137649536, "rewards/total_composite/std": 0.15748561918735504, "reward": 0.8468114137649536, "reward_std": 0.15748561918735504, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12130964547395706, "sampling/sampling_logp_difference/max": 3.5159692764282227, "sampling/importance_sampling_ratio/min": 0.029718982055783272, "sampling/importance_sampling_ratio/mean": 0.996853232383728, "sampling/importance_sampling_ratio/max": 1.968106985092163, "entropy": 0.47702427208423615, "clip_ratio/low_mean": 0.019860627129673958, "clip_ratio/low_min": 0.019860627129673958, "clip_ratio/high_mean": 0.07691173441708088, "clip_ratio/high_max": 0.07691173441708088, "clip_ratio/region_mean": 0.09677236154675484, "reward_total_mean": 0.8468114137649536, "reward_meter_mean": 0.7420197129249573, "reward_meter_std": 0.33649101853370667, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9965255260467529, "reward_repeat_soft_std": 0.007875239476561546, "reward_judge_quality_mean": 0.877500057220459, "reward_judge_quality_std": 0.07869471609592438, "reward_total_composite_mean": 0.8468114137649536, "reward_total_composite_std": 0.15748561918735504} {"timestamp_utc": "2026-04-13T03:37:01Z", "mode": "train", "global_step": 2110, "epoch": 0.2119537920642893, "loss": 0.0049, "grad_norm": 8.562999725341797, "learning_rate": 3.6090909090909093e-06, "num_tokens": 3872927.0, "completions/mean_length": 60.625, "completions/min_length": 55.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.625, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9914700984954834, "rewards/meter/std": 0.007059730123728514, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9706976413726807, "rewards/repeat_soft/std": 0.026481883600354195, "rewards/judge_quality/mean": 0.47999998927116394, "rewards/judge_quality/std": 0.10993505269289017, "rewards/total_composite/mean": 0.8372312784194946, "rewards/total_composite/std": 0.03526733070611954, "reward": 0.8372312784194946, "reward_std": 0.03526733070611954, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13647644221782684, "sampling/sampling_logp_difference/max": 1.0675697326660156, "sampling/importance_sampling_ratio/min": 0.34384313225746155, "sampling/importance_sampling_ratio/mean": 1.0077745914459229, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0265944376587868, "clip_ratio/low_mean": 0.08084983937442303, "clip_ratio/low_min": 0.08084983937442303, "clip_ratio/high_mean": 0.024621212854981422, "clip_ratio/high_max": 0.024621212854981422, "clip_ratio/region_mean": 0.10547105222940445, "reward_total_mean": 0.8372312784194946, "reward_meter_mean": 0.9914700984954834, "reward_meter_std": 0.007059730123728514, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9706976413726807, "reward_repeat_soft_std": 0.026481883600354195, "reward_judge_quality_mean": 0.47999998927116394, "reward_judge_quality_std": 0.10993505269289017, "reward_total_composite_mean": 0.8372312784194946, "reward_total_composite_std": 0.03526733070611954} {"timestamp_utc": "2026-04-13T03:37:08Z", "mode": "train", "global_step": 2111, "epoch": 0.212054244098443, "loss": 0.0081, "grad_norm": 8.372374534606934, "learning_rate": 3.606060606060606e-06, "num_tokens": 3875266.0, "completions/mean_length": 134.375, "completions/min_length": 126.0, "completions/max_length": 146.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 134.375, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 146.0, "rewards/meter/mean": 0.585163950920105, "rewards/meter/std": 0.2980547249317169, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9458664655685425, "rewards/repeat_soft/std": 0.023136859759688377, "rewards/judge_quality/mean": 0.2799999713897705, "rewards/judge_quality/std": 0.09304375946521759, "rewards/total_composite/mean": 0.5234189033508301, "rewards/total_composite/std": 0.2582794427871704, "reward": 0.5234189033508301, "reward_std": 0.2582794427871704, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11218220740556717, "sampling/sampling_logp_difference/max": 1.8493566513061523, "sampling/importance_sampling_ratio/min": 0.15733836591243744, "sampling/importance_sampling_ratio/mean": 1.0177932977676392, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6067464463412762, "clip_ratio/low_mean": 0.02942692907527089, "clip_ratio/low_min": 0.02942692907527089, "clip_ratio/high_mean": 0.07388235488906503, "clip_ratio/high_max": 0.07388235488906503, "clip_ratio/region_mean": 0.10330928396433592, "reward_total_mean": 0.5234189033508301, "reward_meter_mean": 0.585163950920105, "reward_meter_std": 0.2980547249317169, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9458664655685425, "reward_repeat_soft_std": 0.023136859759688377, "reward_judge_quality_mean": 0.2799999713897705, "reward_judge_quality_std": 0.09304375946521759, "reward_total_composite_mean": 0.5234189033508301, "reward_total_composite_std": 0.2582794427871704} {"timestamp_utc": "2026-04-13T03:37:14Z", "mode": "train", "global_step": 2112, "epoch": 0.2121546961325967, "loss": 0.06, "grad_norm": 11.713743209838867, "learning_rate": 3.603030303030303e-06, "num_tokens": 3876945.0, "completions/mean_length": 49.875, "completions/min_length": 42.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.875, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.8719891309738159, "rewards/meter/std": 0.3269916772842407, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9955489635467529, "rewards/repeat_soft/std": 0.005328337196260691, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.8452000021934509, "rewards/total_composite/std": 0.13879762589931488, "reward": 0.8452000021934509, "reward_std": 0.13879764080047607, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14342477917671204, "sampling/sampling_logp_difference/max": 1.9190683364868164, "sampling/importance_sampling_ratio/min": 0.14674361050128937, "sampling/importance_sampling_ratio/mean": 0.9862481355667114, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8669740259647369, "clip_ratio/low_mean": 0.06539422832429409, "clip_ratio/low_min": 0.06539422832429409, "clip_ratio/high_mean": 0.06838103849440813, "clip_ratio/high_max": 0.06838103849440813, "clip_ratio/region_mean": 0.13377526681870222, "reward_total_mean": 0.8452000021934509, "reward_meter_mean": 0.8719891309738159, "reward_meter_std": 0.3269916772842407, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9955489635467529, "reward_repeat_soft_std": 0.005328337196260691, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.8452000021934509, "reward_total_composite_std": 0.13879762589931488} {"timestamp_utc": "2026-04-13T03:37:27Z", "mode": "train", "global_step": 2113, "epoch": 0.21225514816675037, "loss": -0.0299, "grad_norm": 16.93870735168457, "learning_rate": 3.6000000000000003e-06, "num_tokens": 3878425.0, "completions/mean_length": 30.0, "completions/min_length": 26.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.0, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.7397430539131165, "rewards/meter/std": 0.4235769212245941, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9562689065933228, "rewards/repeat_soft/std": 0.015592156909406185, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.817011296749115, "rewards/total_composite/std": 0.22995923459529877, "reward": 0.817011296749115, "reward_std": 0.22995923459529877, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12131988257169724, "sampling/sampling_logp_difference/max": 1.0313239097595215, "sampling/importance_sampling_ratio/min": 0.3565346300601959, "sampling/importance_sampling_ratio/mean": 1.0022146701812744, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7254559025168419, "clip_ratio/low_mean": 0.04313186928629875, "clip_ratio/low_min": 0.04313186928629875, "clip_ratio/high_mean": 0.05372981959953904, "clip_ratio/high_max": 0.05372981959953904, "clip_ratio/region_mean": 0.0968616888858378, "reward_total_mean": 0.817011296749115, "reward_meter_mean": 0.7397430539131165, "reward_meter_std": 0.4235769212245941, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9562689065933228, "reward_repeat_soft_std": 0.015592156909406185, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.817011296749115, "reward_total_composite_std": 0.22995923459529877} {"timestamp_utc": "2026-04-13T03:37:38Z", "mode": "train", "global_step": 2114, "epoch": 0.21235560020090408, "loss": -0.136, "grad_norm": 3.397109031677246, "learning_rate": 3.596969696969697e-06, "num_tokens": 3880312.0, "completions/mean_length": 128.875, "completions/min_length": 67.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 74.14286041259766, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.68594890832901, "rewards/meter/std": 0.3679898977279663, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9773005843162537, "rewards/repeat_soft/std": 0.014944523572921753, "rewards/judge_quality/mean": 0.5362499952316284, "rewards/judge_quality/std": 0.29731839895248413, "rewards/total_composite/mean": 0.6427077054977417, "rewards/total_composite/std": 0.3342384099960327, "reward": 0.6427077054977417, "reward_std": 0.3342384099960327, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11039166897535324, "sampling/sampling_logp_difference/max": 3.2922744750976562, "sampling/importance_sampling_ratio/min": 0.037169214338064194, "sampling/importance_sampling_ratio/mean": 1.0129493474960327, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.41368282586336136, "clip_ratio/low_mean": 0.02076967153698206, "clip_ratio/low_min": 0.02076967153698206, "clip_ratio/high_mean": 0.05763052077963948, "clip_ratio/high_max": 0.05763052077963948, "clip_ratio/region_mean": 0.07840019231662154, "reward_total_mean": 0.6427077054977417, "reward_meter_mean": 0.68594890832901, "reward_meter_std": 0.3679898977279663, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9773005843162537, "reward_repeat_soft_std": 0.014944523572921753, "reward_judge_quality_mean": 0.5362499952316284, "reward_judge_quality_std": 0.29731839895248413, "reward_total_composite_mean": 0.6427077054977417, "reward_total_composite_std": 0.3342384099960327} {"timestamp_utc": "2026-04-13T03:37:45Z", "mode": "train", "global_step": 2115, "epoch": 0.21245605223505776, "loss": 0.0401, "grad_norm": 11.421978950500488, "learning_rate": 3.593939393939394e-06, "num_tokens": 3881964.0, "completions/mean_length": 58.5, "completions/min_length": 45.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.5, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.8008316159248352, "rewards/meter/std": 0.24579596519470215, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9925119876861572, "rewards/repeat_soft/std": 0.007428196258842945, "rewards/judge_quality/mean": 0.7400000095367432, "rewards/judge_quality/std": 0.25707143545150757, "rewards/total_composite/mean": 0.8222503662109375, "rewards/total_composite/std": 0.09261111915111542, "reward": 0.8222503662109375, "reward_std": 0.09261111915111542, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10960116237401962, "sampling/sampling_logp_difference/max": 2.171912670135498, "sampling/importance_sampling_ratio/min": 0.11395943909883499, "sampling/importance_sampling_ratio/mean": 0.9903972744941711, "sampling/importance_sampling_ratio/max": 1.936559796333313, "entropy": 0.6019289828836918, "clip_ratio/low_mean": 0.04765210393816233, "clip_ratio/low_min": 0.04765210393816233, "clip_ratio/high_mean": 0.04800975974649191, "clip_ratio/high_max": 0.04800975974649191, "clip_ratio/region_mean": 0.09566186368465424, "reward_total_mean": 0.8222503662109375, "reward_meter_mean": 0.8008316159248352, "reward_meter_std": 0.24579596519470215, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9925119876861572, "reward_repeat_soft_std": 0.007428196258842945, "reward_judge_quality_mean": 0.7400000095367432, "reward_judge_quality_std": 0.25707143545150757, "reward_total_composite_mean": 0.8222503662109375, "reward_total_composite_std": 0.09261111915111542} {"timestamp_utc": "2026-04-13T03:37:57Z", "mode": "train", "global_step": 2116, "epoch": 0.21255650426921144, "loss": -0.1212, "grad_norm": 2.320309638977051, "learning_rate": 3.590909090909091e-06, "num_tokens": 3883544.0, "completions/mean_length": 168.5, "completions/min_length": 45.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.6381356716156006, "rewards/meter/std": 0.40156200528144836, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9882596731185913, "rewards/repeat_soft/std": 0.011839025653898716, "rewards/judge_quality/mean": 0.5137500166893005, "rewards/judge_quality/std": 0.3617393672466278, "rewards/total_composite/mean": 0.6071173548698425, "rewards/total_composite/std": 0.4053838551044464, "reward": 0.6071173548698425, "reward_std": 0.405383825302124, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13751420378684998, "sampling/sampling_logp_difference/max": 1.5229876041412354, "sampling/importance_sampling_ratio/min": 0.2180594503879547, "sampling/importance_sampling_ratio/mean": 1.03085458278656, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6381911486387253, "clip_ratio/low_mean": 0.01944444514811039, "clip_ratio/low_min": 0.01944444514811039, "clip_ratio/high_mean": 0.0818847930058837, "clip_ratio/high_max": 0.0818847930058837, "clip_ratio/region_mean": 0.10132923815399408, "reward_total_mean": 0.6071173548698425, "reward_meter_mean": 0.6381356716156006, "reward_meter_std": 0.40156200528144836, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9882596731185913, "reward_repeat_soft_std": 0.011839025653898716, "reward_judge_quality_mean": 0.5137500166893005, "reward_judge_quality_std": 0.3617393672466278, "reward_total_composite_mean": 0.6071173548698425, "reward_total_composite_std": 0.4053838551044464} {"timestamp_utc": "2026-04-13T03:38:03Z", "mode": "train", "global_step": 2117, "epoch": 0.21265695630336515, "loss": -0.033, "grad_norm": 7.53915548324585, "learning_rate": 3.587878787878788e-06, "num_tokens": 3885548.0, "completions/mean_length": 66.5, "completions/min_length": 50.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.7392380237579346, "rewards/meter/std": 0.3284810483455658, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.26726123690605164, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9600502848625183, "rewards/repeat_soft/std": 0.05479779839515686, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.6795371174812317, "rewards/total_composite/std": 0.1334114819765091, "reward": 0.6795371174812317, "reward_std": 0.1334114819765091, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12990771234035492, "sampling/sampling_logp_difference/max": 2.624393939971924, "sampling/importance_sampling_ratio/min": 0.07248367369174957, "sampling/importance_sampling_ratio/mean": 1.0012574195861816, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7410685457289219, "clip_ratio/low_mean": 0.05360797233879566, "clip_ratio/low_min": 0.05360797233879566, "clip_ratio/high_mean": 0.08430936932563782, "clip_ratio/high_max": 0.08430936932563782, "clip_ratio/region_mean": 0.13791734166443348, "reward_total_mean": 0.6795371174812317, "reward_meter_mean": 0.7392380237579346, "reward_meter_std": 0.3284810483455658, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.26726123690605164, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9600502848625183, "reward_repeat_soft_std": 0.05479779839515686, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.6795371174812317, "reward_total_composite_std": 0.1334114819765091} {"timestamp_utc": "2026-04-13T03:38:10Z", "mode": "train", "global_step": 2118, "epoch": 0.21275740833751883, "loss": 0.0147, "grad_norm": 13.565587043762207, "learning_rate": 3.584848484848485e-06, "num_tokens": 3887436.0, "completions/mean_length": 55.0, "completions/min_length": 50.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.0, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9957040548324585, "rewards/meter/std": 0.004116056486964226, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9719164371490479, "rewards/repeat_soft/std": 0.02208712324500084, "rewards/judge_quality/mean": 0.7987500429153442, "rewards/judge_quality/std": 0.22465451061725616, "rewards/total_composite/mean": 0.9348834753036499, "rewards/total_composite/std": 0.0692506656050682, "reward": 0.9348834753036499, "reward_std": 0.06925065815448761, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11860547959804535, "sampling/sampling_logp_difference/max": 2.2522566318511963, "sampling/importance_sampling_ratio/min": 0.1051616445183754, "sampling/importance_sampling_ratio/mean": 1.041818380355835, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8821732848882675, "clip_ratio/low_mean": 0.032931034453213215, "clip_ratio/low_min": 0.032931034453213215, "clip_ratio/high_mean": 0.08737270068377256, "clip_ratio/high_max": 0.08737270068377256, "clip_ratio/region_mean": 0.12030373513698578, "reward_total_mean": 0.9348834753036499, "reward_meter_mean": 0.9957040548324585, "reward_meter_std": 0.004116056486964226, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9719164371490479, "reward_repeat_soft_std": 0.02208712324500084, "reward_judge_quality_mean": 0.7987500429153442, "reward_judge_quality_std": 0.22465451061725616, "reward_total_composite_mean": 0.9348834753036499, "reward_total_composite_std": 0.0692506656050682} {"timestamp_utc": "2026-04-13T03:38:17Z", "mode": "train", "global_step": 2119, "epoch": 0.21285786037167254, "loss": 0.0139, "grad_norm": 5.925915241241455, "learning_rate": 3.5818181818181817e-06, "num_tokens": 3889744.0, "completions/mean_length": 112.5, "completions/min_length": 108.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.5, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.8973657488822937, "rewards/meter/std": 0.11911782622337341, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9375796318054199, "rewards/repeat_soft/std": 0.05361942574381828, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.7848225235939026, "rewards/total_composite/std": 0.06924593448638916, "reward": 0.7848225235939026, "reward_std": 0.06924596428871155, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0831569954752922, "sampling/sampling_logp_difference/max": 1.4421414136886597, "sampling/importance_sampling_ratio/min": 0.23642095923423767, "sampling/importance_sampling_ratio/mean": 1.0169012546539307, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4883284494280815, "clip_ratio/low_mean": 0.03674830636009574, "clip_ratio/low_min": 0.03674830636009574, "clip_ratio/high_mean": 0.05801142379641533, "clip_ratio/high_max": 0.05801142379641533, "clip_ratio/region_mean": 0.09475973015651107, "reward_total_mean": 0.7848225235939026, "reward_meter_mean": 0.8973657488822937, "reward_meter_std": 0.11911782622337341, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9375796318054199, "reward_repeat_soft_std": 0.05361942574381828, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.7848225235939026, "reward_total_composite_std": 0.06924593448638916} {"timestamp_utc": "2026-04-13T03:38:23Z", "mode": "train", "global_step": 2120, "epoch": 0.21295831240582622, "loss": 0.0328, "grad_norm": 14.979418754577637, "learning_rate": 3.578787878787879e-06, "num_tokens": 3891363.0, "completions/mean_length": 34.375, "completions/min_length": 31.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.375, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.9340423941612244, "rewards/meter/std": 0.09466143697500229, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9782261252403259, "rewards/repeat_soft/std": 0.011449907906353474, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.7986416816711426, "rewards/total_composite/std": 0.04148436710238457, "reward": 0.7986416816711426, "reward_std": 0.04148437827825546, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11049613356590271, "sampling/sampling_logp_difference/max": 3.301374912261963, "sampling/importance_sampling_ratio/min": 0.03683249279856682, "sampling/importance_sampling_ratio/mean": 0.9831903576850891, "sampling/importance_sampling_ratio/max": 1.536617398262024, "entropy": 0.44999687001109123, "clip_ratio/low_mean": 0.010611205361783504, "clip_ratio/low_min": 0.010611205361783504, "clip_ratio/high_mean": 0.08726473897695541, "clip_ratio/high_max": 0.08726473897695541, "clip_ratio/region_mean": 0.09787594433873892, "reward_total_mean": 0.7986416816711426, "reward_meter_mean": 0.9340423941612244, "reward_meter_std": 0.09466143697500229, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9782261252403259, "reward_repeat_soft_std": 0.011449907906353474, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.7986416816711426, "reward_total_composite_std": 0.04148436710238457} {"timestamp_utc": "2026-04-13T03:38:30Z", "mode": "train", "global_step": 2121, "epoch": 0.2130587644399799, "loss": -0.0199, "grad_norm": 9.934246063232422, "learning_rate": 3.575757575757576e-06, "num_tokens": 3892944.0, "completions/mean_length": 36.625, "completions/min_length": 27.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.625, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.757614254951477, "rewards/meter/std": 0.2781367599964142, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9980446696281433, "rewards/repeat_soft/std": 0.00234316848218441, "rewards/judge_quality/mean": 0.6525000333786011, "rewards/judge_quality/std": 0.2418234497308731, "rewards/total_composite/mean": 0.7771059274673462, "rewards/total_composite/std": 0.12489607185125351, "reward": 0.7771059274673462, "reward_std": 0.12489606440067291, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12987582385540009, "sampling/sampling_logp_difference/max": 2.1490139961242676, "sampling/importance_sampling_ratio/min": 0.11659906804561615, "sampling/importance_sampling_ratio/mean": 1.0057846307754517, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5826064050197601, "clip_ratio/low_mean": 0.05814270209521055, "clip_ratio/low_min": 0.05814270209521055, "clip_ratio/high_mean": 0.08894808031618595, "clip_ratio/high_max": 0.08894808031618595, "clip_ratio/region_mean": 0.1470907824113965, "reward_total_mean": 0.7771059274673462, "reward_meter_mean": 0.757614254951477, "reward_meter_std": 0.2781367599964142, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9980446696281433, "reward_repeat_soft_std": 0.00234316848218441, "reward_judge_quality_mean": 0.6525000333786011, "reward_judge_quality_std": 0.2418234497308731, "reward_total_composite_mean": 0.7771059274673462, "reward_total_composite_std": 0.12489607185125351} {"timestamp_utc": "2026-04-13T03:38:35Z", "mode": "train", "global_step": 2122, "epoch": 0.2131592164741336, "loss": 0.0001, "grad_norm": 9.725767135620117, "learning_rate": 3.5727272727272734e-06, "num_tokens": 3894338.0, "completions/mean_length": 23.25, "completions/min_length": 21.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.25, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.9905533790588379, "rewards/meter/std": 0.01311875693500042, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.5012500286102295, "rewards/judge_quality/std": 0.16974246501922607, "rewards/total_composite/mean": 0.8423740267753601, "rewards/total_composite/std": 0.0519418902695179, "reward": 0.8423740267753601, "reward_std": 0.051941897720098495, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08301753550767899, "sampling/sampling_logp_difference/max": 1.376314401626587, "sampling/importance_sampling_ratio/min": 0.38619691133499146, "sampling/importance_sampling_ratio/mean": 1.0189543962478638, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4913880228996277, "clip_ratio/low_mean": 0.038719207514077425, "clip_ratio/low_min": 0.038719207514077425, "clip_ratio/high_mean": 0.004999999888241291, "clip_ratio/high_max": 0.004999999888241291, "clip_ratio/region_mean": 0.043719207402318716, "reward_total_mean": 0.8423740267753601, "reward_meter_mean": 0.9905533790588379, "reward_meter_std": 0.01311875693500042, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.5012500286102295, "reward_judge_quality_std": 0.16974246501922607, "reward_total_composite_mean": 0.8423740267753601, "reward_total_composite_std": 0.0519418902695179} {"timestamp_utc": "2026-04-13T03:38:42Z", "mode": "train", "global_step": 2123, "epoch": 0.2132596685082873, "loss": 0.0951, "grad_norm": 11.775420188903809, "learning_rate": 3.5696969696969703e-06, "num_tokens": 3896051.0, "completions/mean_length": 55.125, "completions/min_length": 46.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.125, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.5462406873703003, "rewards/meter/std": 0.4103783071041107, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.994552493095398, "rewards/repeat_soft/std": 0.006458591669797897, "rewards/judge_quality/mean": 0.6449999809265137, "rewards/judge_quality/std": 0.24928471446037292, "rewards/total_composite/mean": 0.6793885231018066, "rewards/total_composite/std": 0.19443553686141968, "reward": 0.6793885231018066, "reward_std": 0.19443552196025848, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12754833698272705, "sampling/sampling_logp_difference/max": 1.144587516784668, "sampling/importance_sampling_ratio/min": 0.3183552026748657, "sampling/importance_sampling_ratio/mean": 1.0084306001663208, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7120879143476486, "clip_ratio/low_mean": 0.06090148724615574, "clip_ratio/low_min": 0.06090148724615574, "clip_ratio/high_mean": 0.06419567484408617, "clip_ratio/high_max": 0.06419567484408617, "clip_ratio/region_mean": 0.1250971620902419, "reward_total_mean": 0.6793885231018066, "reward_meter_mean": 0.5462406873703003, "reward_meter_std": 0.4103783071041107, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.994552493095398, "reward_repeat_soft_std": 0.006458591669797897, "reward_judge_quality_mean": 0.6449999809265137, "reward_judge_quality_std": 0.24928471446037292, "reward_total_composite_mean": 0.6793885231018066, "reward_total_composite_std": 0.19443553686141968} {"timestamp_utc": "2026-04-13T03:38:49Z", "mode": "train", "global_step": 2124, "epoch": 0.213360120542441, "loss": -0.009, "grad_norm": 11.564496040344238, "learning_rate": 3.566666666666667e-06, "num_tokens": 3897696.0, "completions/mean_length": 46.625, "completions/min_length": 42.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.625, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9014147520065308, "rewards/meter/std": 0.1965196281671524, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9801899194717407, "rewards/repeat_soft/std": 0.016065942123532295, "rewards/judge_quality/mean": 0.8575000166893005, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.9109055995941162, "rewards/total_composite/std": 0.11049855500459671, "reward": 0.9109055995941162, "reward_std": 0.11049854755401611, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08758105337619781, "sampling/sampling_logp_difference/max": 1.7097606658935547, "sampling/importance_sampling_ratio/min": 0.18090909719467163, "sampling/importance_sampling_ratio/mean": 1.0149306058883667, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5074740685522556, "clip_ratio/low_mean": 0.03679078072309494, "clip_ratio/low_min": 0.03679078072309494, "clip_ratio/high_mean": 0.05592998140491545, "clip_ratio/high_max": 0.05592998140491545, "clip_ratio/region_mean": 0.09272076212801039, "reward_total_mean": 0.9109055995941162, "reward_meter_mean": 0.9014147520065308, "reward_meter_std": 0.1965196281671524, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9801899194717407, "reward_repeat_soft_std": 0.016065942123532295, "reward_judge_quality_mean": 0.8575000166893005, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.9109055995941162, "reward_total_composite_std": 0.11049855500459671} {"timestamp_utc": "2026-04-13T03:38:56Z", "mode": "train", "global_step": 2125, "epoch": 0.21346057257659468, "loss": 0.0545, "grad_norm": 9.002141952514648, "learning_rate": 3.563636363636364e-06, "num_tokens": 3900066.0, "completions/mean_length": 116.25, "completions/min_length": 108.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.25, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.8748704791069031, "rewards/meter/std": 0.2022549957036972, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9860484004020691, "rewards/repeat_soft/std": 0.009179561398923397, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7555465698242188, "rewards/total_composite/std": 0.08574503660202026, "reward": 0.7555465698242188, "reward_std": 0.08574502915143967, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12113501876592636, "sampling/sampling_logp_difference/max": 2.272643804550171, "sampling/importance_sampling_ratio/min": 0.10303940623998642, "sampling/importance_sampling_ratio/mean": 1.0101062059402466, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7100751101970673, "clip_ratio/low_mean": 0.025638440623879433, "clip_ratio/low_min": 0.025638440623879433, "clip_ratio/high_mean": 0.10128554701805115, "clip_ratio/high_max": 0.10128554701805115, "clip_ratio/region_mean": 0.12692398764193058, "reward_total_mean": 0.7555465698242188, "reward_meter_mean": 0.8748704791069031, "reward_meter_std": 0.2022549957036972, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9860484004020691, "reward_repeat_soft_std": 0.009179561398923397, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7555465698242188, "reward_total_composite_std": 0.08574503660202026} {"timestamp_utc": "2026-04-13T03:39:09Z", "mode": "train", "global_step": 2126, "epoch": 0.21356102461074836, "loss": 0.0506, "grad_norm": 11.733000755310059, "learning_rate": 3.560606060606061e-06, "num_tokens": 3901723.0, "completions/mean_length": 49.125, "completions/min_length": 43.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.125, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.7626791000366211, "rewards/meter/std": 0.27298763394355774, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9693494439125061, "rewards/repeat_soft/std": 0.04204733669757843, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.25587037205696106, "rewards/total_composite/mean": 0.773515522480011, "rewards/total_composite/std": 0.16346825659275055, "reward": 0.773515522480011, "reward_std": 0.16346822679042816, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12647971510887146, "sampling/sampling_logp_difference/max": 1.044827938079834, "sampling/importance_sampling_ratio/min": 0.3517523407936096, "sampling/importance_sampling_ratio/mean": 1.0246200561523438, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8503872975707054, "clip_ratio/low_mean": 0.05383660830557346, "clip_ratio/low_min": 0.05383660830557346, "clip_ratio/high_mean": 0.060518407728523016, "clip_ratio/high_max": 0.060518407728523016, "clip_ratio/region_mean": 0.11435501603409648, "reward_total_mean": 0.773515522480011, "reward_meter_mean": 0.7626791000366211, "reward_meter_std": 0.27298763394355774, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9693494439125061, "reward_repeat_soft_std": 0.04204733669757843, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.25587037205696106, "reward_total_composite_mean": 0.773515522480011, "reward_total_composite_std": 0.16346825659275055} {"timestamp_utc": "2026-04-13T03:39:16Z", "mode": "train", "global_step": 2127, "epoch": 0.21366147664490207, "loss": -0.0087, "grad_norm": 12.284749031066895, "learning_rate": 3.557575757575758e-06, "num_tokens": 3903486.0, "completions/mean_length": 52.375, "completions/min_length": 47.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.375, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.7096632719039917, "rewards/meter/std": 0.3018563985824585, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9693694114685059, "rewards/repeat_soft/std": 0.022905606776475906, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.6979104280471802, "rewards/total_composite/std": 0.1347874253988266, "reward": 0.6979104280471802, "reward_std": 0.1347874104976654, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13090239465236664, "sampling/sampling_logp_difference/max": 1.5327863693237305, "sampling/importance_sampling_ratio/min": 0.2159331738948822, "sampling/importance_sampling_ratio/mean": 1.006748080253601, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7586480975151062, "clip_ratio/low_mean": 0.06381674110889435, "clip_ratio/low_min": 0.06381674110889435, "clip_ratio/high_mean": 0.046328675001859665, "clip_ratio/high_max": 0.046328675001859665, "clip_ratio/region_mean": 0.11014541611075401, "reward_total_mean": 0.6979104280471802, "reward_meter_mean": 0.7096632719039917, "reward_meter_std": 0.3018563985824585, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9693694114685059, "reward_repeat_soft_std": 0.022905606776475906, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.6979104280471802, "reward_total_composite_std": 0.1347874253988266} {"timestamp_utc": "2026-04-13T03:39:24Z", "mode": "train", "global_step": 2128, "epoch": 0.21376192867905575, "loss": 0.0223, "grad_norm": 7.285673141479492, "learning_rate": 3.554545454545455e-06, "num_tokens": 3906242.0, "completions/mean_length": 130.5, "completions/min_length": 115.0, "completions/max_length": 142.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 130.5, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 142.0, "rewards/meter/mean": 0.9879094362258911, "rewards/meter/std": 0.010405945591628551, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9752793908119202, "rewards/repeat_soft/std": 0.014630859717726707, "rewards/judge_quality/mean": 0.6200000047683716, "rewards/judge_quality/std": 0.22677870094776154, "rewards/total_composite/mean": 0.8780872225761414, "rewards/total_composite/std": 0.06677749007940292, "reward": 0.8780872225761414, "reward_std": 0.06677747517824173, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11489042639732361, "sampling/sampling_logp_difference/max": 2.1907193660736084, "sampling/importance_sampling_ratio/min": 0.1118362694978714, "sampling/importance_sampling_ratio/mean": 1.0177572965621948, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8030837550759315, "clip_ratio/low_mean": 0.057567525655031204, "clip_ratio/low_min": 0.057567525655031204, "clip_ratio/high_mean": 0.06717841513454914, "clip_ratio/high_max": 0.06717841513454914, "clip_ratio/region_mean": 0.12474594078958035, "reward_total_mean": 0.8780872225761414, "reward_meter_mean": 0.9879094362258911, "reward_meter_std": 0.010405945591628551, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9752793908119202, "reward_repeat_soft_std": 0.014630859717726707, "reward_judge_quality_mean": 0.6200000047683716, "reward_judge_quality_std": 0.22677870094776154, "reward_total_composite_mean": 0.8780872225761414, "reward_total_composite_std": 0.06677749007940292} {"timestamp_utc": "2026-04-13T03:39:31Z", "mode": "train", "global_step": 2129, "epoch": 0.21386238071320945, "loss": 0.0286, "grad_norm": 6.625362396240234, "learning_rate": 3.551515151515152e-06, "num_tokens": 3908590.0, "completions/mean_length": 119.5, "completions/min_length": 110.0, "completions/max_length": 127.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.5, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.9426057934761047, "rewards/meter/std": 0.08890968561172485, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9679479598999023, "rewards/repeat_soft/std": 0.016118047758936882, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.8194674253463745, "rewards/total_composite/std": 0.06493605673313141, "reward": 0.8194674253463745, "reward_std": 0.06493604928255081, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08239325881004333, "sampling/sampling_logp_difference/max": 1.348698377609253, "sampling/importance_sampling_ratio/min": 0.2595779299736023, "sampling/importance_sampling_ratio/mean": 0.996185839176178, "sampling/importance_sampling_ratio/max": 1.85099196434021, "entropy": 0.4357408843934536, "clip_ratio/low_mean": 0.058359991293400526, "clip_ratio/low_min": 0.058359991293400526, "clip_ratio/high_mean": 0.01488483278080821, "clip_ratio/high_max": 0.01488483278080821, "clip_ratio/region_mean": 0.07324482407420874, "reward_total_mean": 0.8194674253463745, "reward_meter_mean": 0.9426057934761047, "reward_meter_std": 0.08890968561172485, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9679479598999023, "reward_repeat_soft_std": 0.016118047758936882, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.8194674253463745, "reward_total_composite_std": 0.06493605673313141} {"timestamp_utc": "2026-04-13T03:39:39Z", "mode": "train", "global_step": 2130, "epoch": 0.21396283274736314, "loss": 0.0054, "grad_norm": 6.893167972564697, "learning_rate": 3.548484848484849e-06, "num_tokens": 3910769.0, "completions/mean_length": 98.375, "completions/min_length": 87.0, "completions/max_length": 115.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.375, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 115.0, "rewards/meter/mean": 0.968575119972229, "rewards/meter/std": 0.030669618397951126, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8836463093757629, "rewards/repeat_soft/std": 0.06788109242916107, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7938483953475952, "rewards/total_composite/std": 0.02365786023437977, "reward": 0.7938483953475952, "reward_std": 0.02365786023437977, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10419881343841553, "sampling/sampling_logp_difference/max": 1.444553256034851, "sampling/importance_sampling_ratio/min": 0.23585142195224762, "sampling/importance_sampling_ratio/mean": 1.0208280086517334, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7422407791018486, "clip_ratio/low_mean": 0.04817429976537824, "clip_ratio/low_min": 0.04817429976537824, "clip_ratio/high_mean": 0.05008118553087115, "clip_ratio/high_max": 0.05008118553087115, "clip_ratio/region_mean": 0.09825548529624939, "reward_total_mean": 0.7938483953475952, "reward_meter_mean": 0.968575119972229, "reward_meter_std": 0.030669618397951126, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8836463093757629, "reward_repeat_soft_std": 0.06788109242916107, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7938483953475952, "reward_total_composite_std": 0.02365786023437977} {"timestamp_utc": "2026-04-13T03:39:51Z", "mode": "train", "global_step": 2131, "epoch": 0.21406328478151682, "loss": -0.2682, "grad_norm": 1.906363606452942, "learning_rate": 3.5454545454545458e-06, "num_tokens": 3913277.0, "completions/mean_length": 255.5, "completions/min_length": 139.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 170.0, "completions/min_terminated_length": 139.0, "completions/max_terminated_length": 187.0, "rewards/meter/mean": 0.9881787300109863, "rewards/meter/std": 0.010883207432925701, "rewards/count_adherence/mean": 0.8500000238418579, "rewards/count_adherence/std": 0.23299294710159302, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9407772421836853, "rewards/repeat_soft/std": 0.047266408801078796, "rewards/judge_quality/mean": 0.2724999785423279, "rewards/judge_quality/std": 0.14119388163089752, "rewards/total_composite/mean": 0.5889289379119873, "rewards/total_composite/std": 0.36470523476600647, "reward": 0.5889289379119873, "reward_std": 0.36470523476600647, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10874069482088089, "sampling/sampling_logp_difference/max": 1.236886978149414, "sampling/importance_sampling_ratio/min": 0.29028648138046265, "sampling/importance_sampling_ratio/mean": 1.0187952518463135, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6435615196824074, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09287768229842186, "clip_ratio/high_max": 0.09287768229842186, "clip_ratio/region_mean": 0.09287768229842186, "reward_total_mean": 0.5889289379119873, "reward_meter_mean": 0.9881787300109863, "reward_meter_std": 0.010883207432925701, "reward_count_adherence_mean": 0.8500000238418579, "reward_count_adherence_std": 0.23299294710159302, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9407772421836853, "reward_repeat_soft_std": 0.047266408801078796, "reward_judge_quality_mean": 0.2724999785423279, "reward_judge_quality_std": 0.14119388163089752, "reward_total_composite_mean": 0.5889289379119873, "reward_total_composite_std": 0.36470523476600647} {"timestamp_utc": "2026-04-13T03:39:57Z", "mode": "train", "global_step": 2132, "epoch": 0.21416373681567052, "loss": -0.0417, "grad_norm": 15.703596115112305, "learning_rate": 3.5424242424242426e-06, "num_tokens": 3914758.0, "completions/mean_length": 29.125, "completions/min_length": 23.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.125, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.7794699668884277, "rewards/meter/std": 0.3799305558204651, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9524038434028625, "rewards/repeat_soft/std": 0.0285562165081501, "rewards/judge_quality/mean": 0.5349999666213989, "rewards/judge_quality/std": 0.24663449823856354, "rewards/total_composite/mean": 0.7565019130706787, "rewards/total_composite/std": 0.14792436361312866, "reward": 0.7565019130706787, "reward_std": 0.14792436361312866, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13615283370018005, "sampling/sampling_logp_difference/max": 1.3782424926757812, "sampling/importance_sampling_ratio/min": 0.25202110409736633, "sampling/importance_sampling_ratio/mean": 1.0458239316940308, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9092608094215393, "clip_ratio/low_mean": 0.03916666656732559, "clip_ratio/low_min": 0.03916666656732559, "clip_ratio/high_mean": 0.10361503530293703, "clip_ratio/high_max": 0.10361503530293703, "clip_ratio/region_mean": 0.14278170187026262, "reward_total_mean": 0.7565019130706787, "reward_meter_mean": 0.7794699668884277, "reward_meter_std": 0.3799305558204651, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9524038434028625, "reward_repeat_soft_std": 0.0285562165081501, "reward_judge_quality_mean": 0.5349999666213989, "reward_judge_quality_std": 0.24663449823856354, "reward_total_composite_mean": 0.7565019130706787, "reward_total_composite_std": 0.14792436361312866} {"timestamp_utc": "2026-04-13T03:40:04Z", "mode": "train", "global_step": 2133, "epoch": 0.2142641888498242, "loss": -0.0796, "grad_norm": 14.671154022216797, "learning_rate": 3.53939393939394e-06, "num_tokens": 3916339.0, "completions/mean_length": 31.625, "completions/min_length": 25.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.625, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.8615745306015015, "rewards/meter/std": 0.34769290685653687, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.6262500286102295, "rewards/judge_quality/std": 0.2432481348514557, "rewards/total_composite/mean": 0.8218334913253784, "rewards/total_composite/std": 0.13825464248657227, "reward": 0.8218334913253784, "reward_std": 0.13825461268424988, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12098155915737152, "sampling/sampling_logp_difference/max": 1.1774439811706543, "sampling/importance_sampling_ratio/min": 0.30806514620780945, "sampling/importance_sampling_ratio/mean": 1.0169662237167358, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6911224946379662, "clip_ratio/low_mean": 0.05154761951416731, "clip_ratio/low_min": 0.05154761951416731, "clip_ratio/high_mean": 0.09338998794555664, "clip_ratio/high_max": 0.09338998794555664, "clip_ratio/region_mean": 0.14493760745972395, "reward_total_mean": 0.8218334913253784, "reward_meter_mean": 0.8615745306015015, "reward_meter_std": 0.34769290685653687, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.6262500286102295, "reward_judge_quality_std": 0.2432481348514557, "reward_total_composite_mean": 0.8218334913253784, "reward_total_composite_std": 0.13825464248657227} {"timestamp_utc": "2026-04-13T03:40:11Z", "mode": "train", "global_step": 2134, "epoch": 0.2143646408839779, "loss": 0.0471, "grad_norm": 6.907683372497559, "learning_rate": 3.5363636363636367e-06, "num_tokens": 3918852.0, "completions/mean_length": 117.125, "completions/min_length": 106.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.125, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.9921795129776001, "rewards/meter/std": 0.004415894392877817, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9579470753669739, "rewards/repeat_soft/std": 0.021282745525240898, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8182754516601562, "rewards/total_composite/std": 0.002720216056331992, "reward": 0.8182754516601562, "reward_std": 0.002720208140090108, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1148286834359169, "sampling/sampling_logp_difference/max": 3.8107123374938965, "sampling/importance_sampling_ratio/min": 0.022132407873868942, "sampling/importance_sampling_ratio/mean": 1.006410002708435, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.628383569419384, "clip_ratio/low_mean": 0.029210069682449102, "clip_ratio/low_min": 0.029210069682449102, "clip_ratio/high_mean": 0.06538395583629608, "clip_ratio/high_max": 0.06538395583629608, "clip_ratio/region_mean": 0.09459402551874518, "reward_total_mean": 0.8182754516601562, "reward_meter_mean": 0.9921795129776001, "reward_meter_std": 0.004415894392877817, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9579470753669739, "reward_repeat_soft_std": 0.021282745525240898, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8182754516601562, "reward_total_composite_std": 0.002720216056331992} {"timestamp_utc": "2026-04-13T03:40:18Z", "mode": "train", "global_step": 2135, "epoch": 0.2144650929181316, "loss": 0.0347, "grad_norm": 10.579270362854004, "learning_rate": 3.5333333333333335e-06, "num_tokens": 3920542.0, "completions/mean_length": 41.25, "completions/min_length": 37.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.25, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.8202252388000488, "rewards/meter/std": 0.25860515236854553, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.990415096282959, "rewards/repeat_soft/std": 0.014687193557620049, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.7651428580284119, "rewards/total_composite/std": 0.1355249136686325, "reward": 0.7651428580284119, "reward_std": 0.1355249136686325, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0927700623869896, "sampling/sampling_logp_difference/max": 1.5458388328552246, "sampling/importance_sampling_ratio/min": 0.2689296007156372, "sampling/importance_sampling_ratio/mean": 1.0207501649856567, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4859706610441208, "clip_ratio/low_mean": 0.03587634768337011, "clip_ratio/low_min": 0.03587634768337011, "clip_ratio/high_mean": 0.06525384332053363, "clip_ratio/high_max": 0.06525384332053363, "clip_ratio/region_mean": 0.10113019100390375, "reward_total_mean": 0.7651428580284119, "reward_meter_mean": 0.8202252388000488, "reward_meter_std": 0.25860515236854553, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.990415096282959, "reward_repeat_soft_std": 0.014687193557620049, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.7651428580284119, "reward_total_composite_std": 0.1355249136686325} {"timestamp_utc": "2026-04-13T03:40:24Z", "mode": "train", "global_step": 2136, "epoch": 0.21456554495228528, "loss": 0.0574, "grad_norm": 11.25467300415039, "learning_rate": 3.5303030303030304e-06, "num_tokens": 3922010.0, "completions/mean_length": 30.5, "completions/min_length": 25.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.5, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.993833065032959, "rewards/meter/std": 0.003033740445971489, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9596524834632874, "rewards/repeat_soft/std": 0.007353283930569887, "rewards/judge_quality/mean": 0.5600000023841858, "rewards/judge_quality/std": 0.22258226573467255, "rewards/total_composite/mean": 0.861190140247345, "rewards/total_composite/std": 0.06671092659235, "reward": 0.861190140247345, "reward_std": 0.06671092659235, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08172979205846786, "sampling/sampling_logp_difference/max": 0.8485732078552246, "sampling/importance_sampling_ratio/min": 0.4280252158641815, "sampling/importance_sampling_ratio/mean": 1.0340964794158936, "sampling/importance_sampling_ratio/max": 1.771716833114624, "entropy": 0.6470238193869591, "clip_ratio/low_mean": 0.09911367669701576, "clip_ratio/low_min": 0.09911367669701576, "clip_ratio/high_mean": 0.017812499776482582, "clip_ratio/high_max": 0.017812499776482582, "clip_ratio/region_mean": 0.11692617647349834, "reward_total_mean": 0.861190140247345, "reward_meter_mean": 0.993833065032959, "reward_meter_std": 0.003033740445971489, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9596524834632874, "reward_repeat_soft_std": 0.007353283930569887, "reward_judge_quality_mean": 0.5600000023841858, "reward_judge_quality_std": 0.22258226573467255, "reward_total_composite_mean": 0.861190140247345, "reward_total_composite_std": 0.06671092659235} {"timestamp_utc": "2026-04-13T03:40:30Z", "mode": "train", "global_step": 2137, "epoch": 0.21466599698643898, "loss": -0.0074, "grad_norm": 16.68299102783203, "learning_rate": 3.5272727272727276e-06, "num_tokens": 3923447.0, "completions/mean_length": 28.625, "completions/min_length": 24.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.625, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9887925386428833, "rewards/meter/std": 0.012463291175663471, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.7362500429153442, "rewards/judge_quality/std": 0.25376805663108826, "rewards/total_composite/mean": 0.9120816588401794, "rewards/total_composite/std": 0.07795832306146622, "reward": 0.9120816588401794, "reward_std": 0.07795832306146622, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10321344435214996, "sampling/sampling_logp_difference/max": 1.5833358764648438, "sampling/importance_sampling_ratio/min": 0.20528914034366608, "sampling/importance_sampling_ratio/mean": 1.013147234916687, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8208130300045013, "clip_ratio/low_mean": 0.05873842537403107, "clip_ratio/low_min": 0.05873842537403107, "clip_ratio/high_mean": 0.0793079910799861, "clip_ratio/high_max": 0.0793079910799861, "clip_ratio/region_mean": 0.13804641645401716, "reward_total_mean": 0.9120816588401794, "reward_meter_mean": 0.9887925386428833, "reward_meter_std": 0.012463291175663471, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.7362500429153442, "reward_judge_quality_std": 0.25376805663108826, "reward_total_composite_mean": 0.9120816588401794, "reward_total_composite_std": 0.07795832306146622} {"timestamp_utc": "2026-04-13T03:40:37Z", "mode": "train", "global_step": 2138, "epoch": 0.21476644902059266, "loss": 0.0022, "grad_norm": 7.168726444244385, "learning_rate": 3.5242424242424244e-06, "num_tokens": 3925830.0, "completions/mean_length": 130.875, "completions/min_length": 123.0, "completions/max_length": 142.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 130.875, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 142.0, "rewards/meter/mean": 0.764907956123352, "rewards/meter/std": 0.31190475821495056, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9777035713195801, "rewards/repeat_soft/std": 0.011720172129571438, "rewards/judge_quality/mean": 0.6200000047683716, "rewards/judge_quality/std": 0.16903087496757507, "rewards/total_composite/mean": 0.7779788970947266, "rewards/total_composite/std": 0.14218224585056305, "reward": 0.7779788970947266, "reward_std": 0.14218223094940186, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10287867486476898, "sampling/sampling_logp_difference/max": 1.9923405647277832, "sampling/importance_sampling_ratio/min": 0.13637585937976837, "sampling/importance_sampling_ratio/mean": 1.0113550424575806, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5727556087076664, "clip_ratio/low_mean": 0.04083974100649357, "clip_ratio/low_min": 0.04083974100649357, "clip_ratio/high_mean": 0.057483539916574955, "clip_ratio/high_max": 0.057483539916574955, "clip_ratio/region_mean": 0.09832328092306852, "reward_total_mean": 0.7779788970947266, "reward_meter_mean": 0.764907956123352, "reward_meter_std": 0.31190475821495056, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9777035713195801, "reward_repeat_soft_std": 0.011720172129571438, "reward_judge_quality_mean": 0.6200000047683716, "reward_judge_quality_std": 0.16903087496757507, "reward_total_composite_mean": 0.7779788970947266, "reward_total_composite_std": 0.14218224585056305} {"timestamp_utc": "2026-04-13T03:40:43Z", "mode": "train", "global_step": 2139, "epoch": 0.21486690105474635, "loss": 0.1045, "grad_norm": 12.498177528381348, "learning_rate": 3.5212121212121213e-06, "num_tokens": 3927171.0, "completions/mean_length": 29.625, "completions/min_length": 27.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.625, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9801185131072998, "rewards/meter/std": 0.024261321872472763, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9580697417259216, "rewards/repeat_soft/std": 0.012530574575066566, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.816235363483429, "rewards/total_composite/std": 0.013692312873899937, "reward": 0.816235363483429, "reward_std": 0.013692302629351616, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10321921855211258, "sampling/sampling_logp_difference/max": 1.2386038303375244, "sampling/importance_sampling_ratio/min": 0.3592245876789093, "sampling/importance_sampling_ratio/mean": 1.025154709815979, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6337702386081219, "clip_ratio/low_mean": 0.03405359108000994, "clip_ratio/low_min": 0.03405359108000994, "clip_ratio/high_mean": 0.049007938243448734, "clip_ratio/high_max": 0.049007938243448734, "clip_ratio/region_mean": 0.08306152932345867, "reward_total_mean": 0.816235363483429, "reward_meter_mean": 0.9801185131072998, "reward_meter_std": 0.024261321872472763, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9580697417259216, "reward_repeat_soft_std": 0.012530574575066566, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.816235363483429, "reward_total_composite_std": 0.013692312873899937} {"timestamp_utc": "2026-04-13T03:40:50Z", "mode": "train", "global_step": 2140, "epoch": 0.21496735308890005, "loss": 0.0327, "grad_norm": 24.728294372558594, "learning_rate": 3.5181818181818185e-06, "num_tokens": 3928777.0, "completions/mean_length": 38.75, "completions/min_length": 30.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.75, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.913447380065918, "rewards/meter/std": 0.10668312758207321, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9805880784988403, "rewards/repeat_soft/std": 0.01584444008767605, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7918601632118225, "rewards/total_composite/std": 0.04778134450316429, "reward": 0.7918601632118225, "reward_std": 0.04778135195374489, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.140176460146904, "sampling/sampling_logp_difference/max": 1.7722889184951782, "sampling/importance_sampling_ratio/min": 0.16994355618953705, "sampling/importance_sampling_ratio/mean": 0.9840584993362427, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5775543563067913, "clip_ratio/low_mean": 0.009868420660495758, "clip_ratio/low_min": 0.009868420660495758, "clip_ratio/high_mean": 0.10992123000323772, "clip_ratio/high_max": 0.10992123000323772, "clip_ratio/region_mean": 0.11978965066373348, "reward_total_mean": 0.7918601632118225, "reward_meter_mean": 0.913447380065918, "reward_meter_std": 0.10668312758207321, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9805880784988403, "reward_repeat_soft_std": 0.01584444008767605, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7918601632118225, "reward_total_composite_std": 0.04778134450316429} {"timestamp_utc": "2026-04-13T03:40:56Z", "mode": "train", "global_step": 2141, "epoch": 0.21506780512305373, "loss": 0.0058, "grad_norm": 15.577269554138184, "learning_rate": 3.5151515151515154e-06, "num_tokens": 3930490.0, "completions/mean_length": 58.125, "completions/min_length": 51.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.125, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.8945577144622803, "rewards/meter/std": 0.2582120895385742, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9888720512390137, "rewards/repeat_soft/std": 0.009211272932589054, "rewards/judge_quality/mean": 0.6637499928474426, "rewards/judge_quality/std": 0.2351861298084259, "rewards/total_composite/mean": 0.8505631685256958, "rewards/total_composite/std": 0.10642503201961517, "reward": 0.8505631685256958, "reward_std": 0.10642502456903458, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10762235522270203, "sampling/sampling_logp_difference/max": 1.838700532913208, "sampling/importance_sampling_ratio/min": 0.1590239405632019, "sampling/importance_sampling_ratio/mean": 1.0276066064834595, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.710274338722229, "clip_ratio/low_mean": 0.06474559707567096, "clip_ratio/low_min": 0.06474559707567096, "clip_ratio/high_mean": 0.027594097424298525, "clip_ratio/high_max": 0.027594097424298525, "clip_ratio/region_mean": 0.09233969449996948, "reward_total_mean": 0.8505631685256958, "reward_meter_mean": 0.8945577144622803, "reward_meter_std": 0.2582120895385742, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9888720512390137, "reward_repeat_soft_std": 0.009211272932589054, "reward_judge_quality_mean": 0.6637499928474426, "reward_judge_quality_std": 0.2351861298084259, "reward_total_composite_mean": 0.8505631685256958, "reward_total_composite_std": 0.10642503201961517} {"timestamp_utc": "2026-04-13T03:41:03Z", "mode": "train", "global_step": 2142, "epoch": 0.21516825715720744, "loss": 0.1006, "grad_norm": 12.33377742767334, "learning_rate": 3.512121212121212e-06, "num_tokens": 3932556.0, "completions/mean_length": 100.25, "completions/min_length": 84.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.25, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.5330076813697815, "rewards/meter/std": 0.3587382435798645, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9687686562538147, "rewards/repeat_soft/std": 0.02440960332751274, "rewards/judge_quality/mean": 0.6825000047683716, "rewards/judge_quality/std": 0.23260945081710815, "rewards/total_composite/mean": 0.6914803385734558, "rewards/total_composite/std": 0.16618722677230835, "reward": 0.6914803385734558, "reward_std": 0.16618722677230835, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09807722270488739, "sampling/sampling_logp_difference/max": 1.4101365804672241, "sampling/importance_sampling_ratio/min": 0.24410995841026306, "sampling/importance_sampling_ratio/mean": 1.0070888996124268, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5894207991659641, "clip_ratio/low_mean": 0.027401371393352747, "clip_ratio/low_min": 0.027401371393352747, "clip_ratio/high_mean": 0.059058858547359705, "clip_ratio/high_max": 0.059058858547359705, "clip_ratio/region_mean": 0.08646022994071245, "reward_total_mean": 0.6914803385734558, "reward_meter_mean": 0.5330076813697815, "reward_meter_std": 0.3587382435798645, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9687686562538147, "reward_repeat_soft_std": 0.02440960332751274, "reward_judge_quality_mean": 0.6825000047683716, "reward_judge_quality_std": 0.23260945081710815, "reward_total_composite_mean": 0.6914803385734558, "reward_total_composite_std": 0.16618722677230835} {"timestamp_utc": "2026-04-13T03:41:10Z", "mode": "train", "global_step": 2143, "epoch": 0.21526870919136112, "loss": -0.001, "grad_norm": 8.933420181274414, "learning_rate": 3.509090909090909e-06, "num_tokens": 3934519.0, "completions/mean_length": 55.375, "completions/min_length": 51.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.375, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9942647218704224, "rewards/meter/std": 0.0026977756060659885, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9866456985473633, "rewards/repeat_soft/std": 0.017105858772993088, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.824333667755127, "rewards/total_composite/std": 0.004998964257538319, "reward": 0.824333667755127, "reward_std": 0.004998969379812479, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10806433111429214, "sampling/sampling_logp_difference/max": 1.819555640220642, "sampling/importance_sampling_ratio/min": 0.16209776699543, "sampling/importance_sampling_ratio/mean": 1.0010638236999512, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6283224821090698, "clip_ratio/low_mean": 0.0540948580019176, "clip_ratio/low_min": 0.0540948580019176, "clip_ratio/high_mean": 0.035933840088546276, "clip_ratio/high_max": 0.035933840088546276, "clip_ratio/region_mean": 0.09002869809046388, "reward_total_mean": 0.824333667755127, "reward_meter_mean": 0.9942647218704224, "reward_meter_std": 0.0026977756060659885, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9866456985473633, "reward_repeat_soft_std": 0.017105858772993088, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.824333667755127, "reward_total_composite_std": 0.004998964257538319} {"timestamp_utc": "2026-04-13T03:41:17Z", "mode": "train", "global_step": 2144, "epoch": 0.2153691612255148, "loss": 0.0426, "grad_norm": 11.139668464660645, "learning_rate": 3.5060606060606063e-06, "num_tokens": 3936822.0, "completions/mean_length": 98.875, "completions/min_length": 91.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.875, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.9830113649368286, "rewards/meter/std": 0.016836389899253845, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9587432146072388, "rewards/repeat_soft/std": 0.025728648528456688, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.807854413986206, "rewards/total_composite/std": 0.020834308117628098, "reward": 0.807854413986206, "reward_std": 0.02083432301878929, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12403535097837448, "sampling/sampling_logp_difference/max": 1.5484628677368164, "sampling/importance_sampling_ratio/min": 0.21257446706295013, "sampling/importance_sampling_ratio/mean": 1.0011252164840698, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7127158865332603, "clip_ratio/low_mean": 0.022320540621876717, "clip_ratio/low_min": 0.022320540621876717, "clip_ratio/high_mean": 0.0857872711494565, "clip_ratio/high_max": 0.0857872711494565, "clip_ratio/region_mean": 0.10810781177133322, "reward_total_mean": 0.807854413986206, "reward_meter_mean": 0.9830113649368286, "reward_meter_std": 0.016836389899253845, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9587432146072388, "reward_repeat_soft_std": 0.025728648528456688, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.807854413986206, "reward_total_composite_std": 0.020834308117628098} {"timestamp_utc": "2026-04-13T03:41:24Z", "mode": "train", "global_step": 2145, "epoch": 0.2154696132596685, "loss": -0.029, "grad_norm": 14.630890846252441, "learning_rate": 3.503030303030303e-06, "num_tokens": 3938425.0, "completions/mean_length": 32.375, "completions/min_length": 25.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.375, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9865705966949463, "rewards/meter/std": 0.015708118677139282, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9505681991577148, "rewards/repeat_soft/std": 0.03374826908111572, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.8382635712623596, "rewards/total_composite/std": 0.04936946928501129, "reward": 0.8382635712623596, "reward_std": 0.04936946555972099, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13054130971431732, "sampling/sampling_logp_difference/max": 1.1259181499481201, "sampling/importance_sampling_ratio/min": 0.32435449957847595, "sampling/importance_sampling_ratio/mean": 1.0311074256896973, "sampling/importance_sampling_ratio/max": 1.8700436353683472, "entropy": 1.197226032614708, "clip_ratio/low_mean": 0.08470756001770496, "clip_ratio/low_min": 0.08470756001770496, "clip_ratio/high_mean": 0.01689189113676548, "clip_ratio/high_max": 0.01689189113676548, "clip_ratio/region_mean": 0.10159945115447044, "reward_total_mean": 0.8382635712623596, "reward_meter_mean": 0.9865705966949463, "reward_meter_std": 0.015708118677139282, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9505681991577148, "reward_repeat_soft_std": 0.03374826908111572, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.8382635712623596, "reward_total_composite_std": 0.04936946928501129} {"timestamp_utc": "2026-04-13T03:41:32Z", "mode": "train", "global_step": 2146, "epoch": 0.2155700652938222, "loss": 0.0184, "grad_norm": 6.129201889038086, "learning_rate": 3.5e-06, "num_tokens": 3941128.0, "completions/mean_length": 164.875, "completions/min_length": 158.0, "completions/max_length": 180.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 164.875, "completions/min_terminated_length": 158.0, "completions/max_terminated_length": 180.0, "rewards/meter/mean": 0.9805617928504944, "rewards/meter/std": 0.020081989467144012, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8554741144180298, "rewards/repeat_soft/std": 0.04711994528770447, "rewards/judge_quality/mean": 0.3137499690055847, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7709252238273621, "rewards/total_composite/std": 0.030566910281777382, "reward": 0.7709252238273621, "reward_std": 0.030566908419132233, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10743807256221771, "sampling/sampling_logp_difference/max": 2.412043571472168, "sampling/importance_sampling_ratio/min": 0.08963193744421005, "sampling/importance_sampling_ratio/mean": 1.006289005279541, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.612006776034832, "clip_ratio/low_mean": 0.06911618169397116, "clip_ratio/low_min": 0.06911618169397116, "clip_ratio/high_mean": 0.0311793964356184, "clip_ratio/high_max": 0.0311793964356184, "clip_ratio/region_mean": 0.10029557812958956, "reward_total_mean": 0.7709252238273621, "reward_meter_mean": 0.9805617928504944, "reward_meter_std": 0.020081989467144012, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8554741144180298, "reward_repeat_soft_std": 0.04711994528770447, "reward_judge_quality_mean": 0.3137499690055847, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7709252238273621, "reward_total_composite_std": 0.030566910281777382} {"timestamp_utc": "2026-04-13T03:41:38Z", "mode": "train", "global_step": 2147, "epoch": 0.2156705173279759, "loss": 0.0058, "grad_norm": 9.418452262878418, "learning_rate": 3.496969696969697e-06, "num_tokens": 3942972.0, "completions/mean_length": 56.5, "completions/min_length": 52.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.5, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.8744560480117798, "rewards/meter/std": 0.14824706315994263, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9771402478218079, "rewards/repeat_soft/std": 0.01996496133506298, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.7642192244529724, "rewards/total_composite/std": 0.06529989838600159, "reward": 0.7642192244529724, "reward_std": 0.06529989838600159, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1277335286140442, "sampling/sampling_logp_difference/max": 1.0189635753631592, "sampling/importance_sampling_ratio/min": 0.3609688878059387, "sampling/importance_sampling_ratio/mean": 1.0278067588806152, "sampling/importance_sampling_ratio/max": 1.956162452697754, "entropy": 0.9015491232275963, "clip_ratio/low_mean": 0.05576263600960374, "clip_ratio/low_min": 0.05576263600960374, "clip_ratio/high_mean": 0.051250058226287365, "clip_ratio/high_max": 0.051250058226287365, "clip_ratio/region_mean": 0.1070126942358911, "reward_total_mean": 0.7642192244529724, "reward_meter_mean": 0.8744560480117798, "reward_meter_std": 0.14824706315994263, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9771402478218079, "reward_repeat_soft_std": 0.01996496133506298, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.7642192244529724, "reward_total_composite_std": 0.06529989838600159} {"timestamp_utc": "2026-04-13T03:41:45Z", "mode": "train", "global_step": 2148, "epoch": 0.21577096936212958, "loss": -0.0125, "grad_norm": 8.189531326293945, "learning_rate": 3.493939393939394e-06, "num_tokens": 3945034.0, "completions/mean_length": 77.75, "completions/min_length": 64.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.75, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.9607934951782227, "rewards/meter/std": 0.029615148901939392, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9934859275817871, "rewards/repeat_soft/std": 0.008437142707407475, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.8239556550979614, "rewards/total_composite/std": 0.04043324291706085, "reward": 0.8239556550979614, "reward_std": 0.04043324291706085, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12966513633728027, "sampling/sampling_logp_difference/max": 4.731259822845459, "sampling/importance_sampling_ratio/min": 0.008815358392894268, "sampling/importance_sampling_ratio/mean": 1.0034297704696655, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6838011890649796, "clip_ratio/low_mean": 0.08887037448585033, "clip_ratio/low_min": 0.08887037448585033, "clip_ratio/high_mean": 0.030570652335882187, "clip_ratio/high_max": 0.030570652335882187, "clip_ratio/region_mean": 0.11944102682173252, "reward_total_mean": 0.8239556550979614, "reward_meter_mean": 0.9607934951782227, "reward_meter_std": 0.029615148901939392, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9934859275817871, "reward_repeat_soft_std": 0.008437142707407475, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.8239556550979614, "reward_total_composite_std": 0.04043324291706085} {"timestamp_utc": "2026-04-13T03:41:53Z", "mode": "train", "global_step": 2149, "epoch": 0.21587142139628326, "loss": -0.0047, "grad_norm": 11.9099760055542, "learning_rate": 3.4909090909090913e-06, "num_tokens": 3946726.0, "completions/mean_length": 52.5, "completions/min_length": 49.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.5, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.984458327293396, "rewards/meter/std": 0.005085899028927088, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9740910530090332, "rewards/repeat_soft/std": 0.0188217181712389, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.8111653327941895, "rewards/total_composite/std": 0.020054355263710022, "reward": 0.8111653327941895, "reward_std": 0.020054353401064873, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1129683405160904, "sampling/sampling_logp_difference/max": 1.3283252716064453, "sampling/importance_sampling_ratio/min": 0.26492056250572205, "sampling/importance_sampling_ratio/mean": 1.001664638519287, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6872755363583565, "clip_ratio/low_mean": 0.01225490216165781, "clip_ratio/low_min": 0.01225490216165781, "clip_ratio/high_mean": 0.11573630664497614, "clip_ratio/high_max": 0.11573630664497614, "clip_ratio/region_mean": 0.12799120880663395, "reward_total_mean": 0.8111653327941895, "reward_meter_mean": 0.984458327293396, "reward_meter_std": 0.005085899028927088, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9740910530090332, "reward_repeat_soft_std": 0.0188217181712389, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.8111653327941895, "reward_total_composite_std": 0.020054355263710022} {"timestamp_utc": "2026-04-13T03:41:59Z", "mode": "train", "global_step": 2150, "epoch": 0.21597187343043697, "loss": 0.0486, "grad_norm": 12.220235824584961, "learning_rate": 3.4878787878787885e-06, "num_tokens": 3948718.0, "completions/mean_length": 59.0, "completions/min_length": 54.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.7072016000747681, "rewards/meter/std": 0.3522599935531616, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9957247972488403, "rewards/repeat_soft/std": 0.00740971090272069, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.7125632166862488, "rewards/total_composite/std": 0.1828254759311676, "reward": 0.7125632166862488, "reward_std": 0.1828254759311676, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13014347851276398, "sampling/sampling_logp_difference/max": 1.3353166580200195, "sampling/importance_sampling_ratio/min": 0.2630748450756073, "sampling/importance_sampling_ratio/mean": 0.9924003481864929, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.660081259906292, "clip_ratio/low_mean": 0.028502383269369602, "clip_ratio/low_min": 0.028502383269369602, "clip_ratio/high_mean": 0.09667748399078846, "clip_ratio/high_max": 0.09667748399078846, "clip_ratio/region_mean": 0.12517986726015806, "reward_total_mean": 0.7125632166862488, "reward_meter_mean": 0.7072016000747681, "reward_meter_std": 0.3522599935531616, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9957247972488403, "reward_repeat_soft_std": 0.00740971090272069, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.7125632166862488, "reward_total_composite_std": 0.1828254759311676} {"timestamp_utc": "2026-04-13T03:42:56Z", "mode": "eval", "global_step": 2150, "epoch": 0.21597187343043697, "eval_loss": NaN, "eval_runtime": 56.0056, "eval_samples_per_second": 1.428, "eval_steps_per_second": 0.179, "eval_num_tokens": 3948718.0, "eval_completions/mean_length": 99.025, "eval_completions/min_length": 38.4, "eval_completions/max_length": 232.5, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 88.46250114440917, "eval_completions/min_terminated_length": 38.4, "eval_completions/max_terminated_length": 157.8, "eval_rewards/meter/mean": 0.8225147724151611, "eval_rewards/meter/std": 0.23536320626735688, "eval_rewards/count_adherence/mean": 0.9722916662693024, "eval_rewards/count_adherence/std": 0.07045509256422519, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.07071067690849304, "eval_rewards/repeat_soft/mean": 0.9479248821735382, "eval_rewards/repeat_soft/std": 0.05033034216612577, "eval_rewards/judge_quality/mean": 0.40524998903274534, "eval_rewards/judge_quality/std": 0.11583352722227573, "eval_rewards/total_composite/mean": 0.7218496084213257, "eval_rewards/total_composite/std": 0.1410735823214054, "eval_reward": 0.7218496084213257, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.05997037552297115, "eval_sampling/sampling_logp_difference/max": 1.0490092754364013, "eval_sampling/importance_sampling_ratio/min": 0.3544620335102081, "eval_sampling/importance_sampling_ratio/mean": 1.0117972016334533, "eval_sampling/importance_sampling_ratio/max": 1.4437299847602845, "eval_entropy": 0.6626357197761535, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7218496084213257, "eval_reward_meter_mean": 0.8225147724151611, "eval_reward_meter_std": 0.23536320626735688, "eval_reward_count_adherence_mean": 0.9722916662693024, "eval_reward_count_adherence_std": 0.07045509256422519, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.07071067690849304, "eval_reward_repeat_soft_mean": 0.9479248821735382, "eval_reward_repeat_soft_std": 0.05033034216612577, "eval_reward_judge_quality_mean": 0.40524998903274534, "eval_reward_judge_quality_std": 0.11583352722227573, "eval_reward_total_composite_mean": 0.7218496084213257, "eval_reward_total_composite_std": 0.1410735823214054} {"timestamp_utc": "2026-04-13T03:43:07Z", "mode": "train", "global_step": 2151, "epoch": 0.21607232546459065, "loss": 0.0359, "grad_norm": 9.714478492736816, "learning_rate": 3.4848484848484854e-06, "num_tokens": 3951169.0, "completions/mean_length": 108.375, "completions/min_length": 102.0, "completions/max_length": 119.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.375, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.9427931308746338, "rewards/meter/std": 0.08897273987531662, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9642243385314941, "rewards/repeat_soft/std": 0.03727663680911064, "rewards/judge_quality/mean": 0.5362499952316284, "rewards/judge_quality/std": 0.22385822236537933, "rewards/total_composite/mean": 0.8315543532371521, "rewards/total_composite/std": 0.08598554879426956, "reward": 0.8315543532371521, "reward_std": 0.08598554134368896, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11962401121854782, "sampling/sampling_logp_difference/max": 1.6294617652893066, "sampling/importance_sampling_ratio/min": 0.19603504240512848, "sampling/importance_sampling_ratio/mean": 1.0121458768844604, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.526893425732851, "clip_ratio/low_mean": 0.06763226725161076, "clip_ratio/low_min": 0.06763226725161076, "clip_ratio/high_mean": 0.049737393856048584, "clip_ratio/high_max": 0.049737393856048584, "clip_ratio/region_mean": 0.11736966110765934, "reward_total_mean": 0.8315543532371521, "reward_meter_mean": 0.9427931308746338, "reward_meter_std": 0.08897273987531662, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9642243385314941, "reward_repeat_soft_std": 0.03727663680911064, "reward_judge_quality_mean": 0.5362499952316284, "reward_judge_quality_std": 0.22385822236537933, "reward_total_composite_mean": 0.8315543532371521, "reward_total_composite_std": 0.08598554879426956} {"timestamp_utc": "2026-04-13T03:43:15Z", "mode": "train", "global_step": 2152, "epoch": 0.21617277749874436, "loss": 0.0555, "grad_norm": 7.900693893432617, "learning_rate": 3.481818181818182e-06, "num_tokens": 3953788.0, "completions/mean_length": 136.375, "completions/min_length": 125.0, "completions/max_length": 154.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.375, "completions/min_terminated_length": 125.0, "completions/max_terminated_length": 154.0, "rewards/meter/mean": 0.8567699193954468, "rewards/meter/std": 0.18241705000400543, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9221246242523193, "rewards/repeat_soft/std": 0.053475964814424515, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7346339225769043, "rewards/total_composite/std": 0.06814753264188766, "reward": 0.7346339225769043, "reward_std": 0.06814754754304886, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11076714843511581, "sampling/sampling_logp_difference/max": 3.4605698585510254, "sampling/importance_sampling_ratio/min": 0.03141185641288757, "sampling/importance_sampling_ratio/mean": 0.9958487153053284, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5657911412417889, "clip_ratio/low_mean": 0.01696654036641121, "clip_ratio/low_min": 0.01696654036641121, "clip_ratio/high_mean": 0.07202211767435074, "clip_ratio/high_max": 0.07202211767435074, "clip_ratio/region_mean": 0.08898865804076195, "reward_total_mean": 0.7346339225769043, "reward_meter_mean": 0.8567699193954468, "reward_meter_std": 0.18241705000400543, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9221246242523193, "reward_repeat_soft_std": 0.053475964814424515, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7346339225769043, "reward_total_composite_std": 0.06814753264188766} {"timestamp_utc": "2026-04-13T03:43:25Z", "mode": "train", "global_step": 2153, "epoch": 0.21627322953289804, "loss": 0.0141, "grad_norm": 11.902151107788086, "learning_rate": 3.4787878787878795e-06, "num_tokens": 3955493.0, "completions/mean_length": 63.125, "completions/min_length": 47.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.125, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.6630653142929077, "rewards/meter/std": 0.3864939510822296, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9587947726249695, "rewards/repeat_soft/std": 0.05369778349995613, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.6725088357925415, "rewards/total_composite/std": 0.17040421068668365, "reward": 0.6725088357925415, "reward_std": 0.17040421068668365, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09859964996576309, "sampling/sampling_logp_difference/max": 1.3085227012634277, "sampling/importance_sampling_ratio/min": 0.27021896839141846, "sampling/importance_sampling_ratio/mean": 1.0123451948165894, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5613971725106239, "clip_ratio/low_mean": 0.03743140306323767, "clip_ratio/low_min": 0.03743140306323767, "clip_ratio/high_mean": 0.06932028196752071, "clip_ratio/high_max": 0.06932028196752071, "clip_ratio/region_mean": 0.10675168503075838, "reward_total_mean": 0.6725088357925415, "reward_meter_mean": 0.6630653142929077, "reward_meter_std": 0.3864939510822296, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9587947726249695, "reward_repeat_soft_std": 0.05369778349995613, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.6725088357925415, "reward_total_composite_std": 0.17040421068668365} {"timestamp_utc": "2026-04-13T03:43:32Z", "mode": "train", "global_step": 2154, "epoch": 0.21637368156705172, "loss": 0.004, "grad_norm": 9.865124702453613, "learning_rate": 3.4757575757575763e-06, "num_tokens": 3957095.0, "completions/mean_length": 52.25, "completions/min_length": 44.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.25, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9501558542251587, "rewards/meter/std": 0.054502688348293304, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9943901896476746, "rewards/repeat_soft/std": 0.005734652746468782, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.8228841423988342, "rewards/total_composite/std": 0.032150380313396454, "reward": 0.8228841423988342, "reward_std": 0.03215039148926735, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10806025564670563, "sampling/sampling_logp_difference/max": 1.3990569114685059, "sampling/importance_sampling_ratio/min": 0.24682964384555817, "sampling/importance_sampling_ratio/mean": 1.019521713256836, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6422567293047905, "clip_ratio/low_mean": 0.09498598147183657, "clip_ratio/low_min": 0.09498598147183657, "clip_ratio/high_mean": 0.029431217350065708, "clip_ratio/high_max": 0.029431217350065708, "clip_ratio/region_mean": 0.12441719882190228, "reward_total_mean": 0.8228841423988342, "reward_meter_mean": 0.9501558542251587, "reward_meter_std": 0.054502688348293304, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9943901896476746, "reward_repeat_soft_std": 0.005734652746468782, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.8228841423988342, "reward_total_composite_std": 0.032150380313396454} {"timestamp_utc": "2026-04-13T03:43:39Z", "mode": "train", "global_step": 2155, "epoch": 0.21647413360120543, "loss": 0.0454, "grad_norm": 7.9287333488464355, "learning_rate": 3.472727272727273e-06, "num_tokens": 3959089.0, "completions/mean_length": 66.25, "completions/min_length": 56.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.25, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.8408258557319641, "rewards/meter/std": 0.26814213395118713, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9611886739730835, "rewards/repeat_soft/std": 0.03553975373506546, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7453655004501343, "rewards/total_composite/std": 0.11913066357374191, "reward": 0.7453655004501343, "reward_std": 0.11913067102432251, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10546521097421646, "sampling/sampling_logp_difference/max": 1.3075153827667236, "sampling/importance_sampling_ratio/min": 0.2704913020133972, "sampling/importance_sampling_ratio/mean": 1.0139678716659546, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5396451130509377, "clip_ratio/low_mean": 0.029273715801537037, "clip_ratio/low_min": 0.029273715801537037, "clip_ratio/high_mean": 0.07048653857782483, "clip_ratio/high_max": 0.07048653857782483, "clip_ratio/region_mean": 0.09976025437936187, "reward_total_mean": 0.7453655004501343, "reward_meter_mean": 0.8408258557319641, "reward_meter_std": 0.26814213395118713, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9611886739730835, "reward_repeat_soft_std": 0.03553975373506546, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7453655004501343, "reward_total_composite_std": 0.11913066357374191} {"timestamp_utc": "2026-04-13T03:43:50Z", "mode": "train", "global_step": 2156, "epoch": 0.2165745856353591, "loss": 0.0524, "grad_norm": 14.556824684143066, "learning_rate": 3.46969696969697e-06, "num_tokens": 3960809.0, "completions/mean_length": 63.0, "completions/min_length": 56.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.8611014485359192, "rewards/meter/std": 0.3050040900707245, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9700068235397339, "rewards/repeat_soft/std": 0.02426403947174549, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7661212682723999, "rewards/total_composite/std": 0.1360470950603485, "reward": 0.7661212682723999, "reward_std": 0.13604708015918732, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12786293029785156, "sampling/sampling_logp_difference/max": 1.8063380718231201, "sampling/importance_sampling_ratio/min": 0.16425451636314392, "sampling/importance_sampling_ratio/mean": 0.9972307682037354, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7450784221291542, "clip_ratio/low_mean": 0.010869565419852734, "clip_ratio/low_min": 0.010869565419852734, "clip_ratio/high_mean": 0.1240364545956254, "clip_ratio/high_max": 0.1240364545956254, "clip_ratio/region_mean": 0.13490602001547813, "reward_total_mean": 0.7661212682723999, "reward_meter_mean": 0.8611014485359192, "reward_meter_std": 0.3050040900707245, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9700068235397339, "reward_repeat_soft_std": 0.02426403947174549, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7661212682723999, "reward_total_composite_std": 0.1360470950603485} {"timestamp_utc": "2026-04-13T03:44:02Z", "mode": "train", "global_step": 2157, "epoch": 0.21667503766951282, "loss": -0.1508, "grad_norm": 3.275866746902466, "learning_rate": 3.4666666666666672e-06, "num_tokens": 3962767.0, "completions/mean_length": 141.75, "completions/min_length": 79.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 88.85714721679688, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.3656251132488251, "rewards/meter/std": 0.24074342846870422, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9848576784133911, "rewards/repeat_soft/std": 0.006769011728465557, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19171687960624695, "rewards/total_composite/mean": 0.5033431053161621, "rewards/total_composite/std": 0.2377743273973465, "reward": 0.5033431053161621, "reward_std": 0.2377743273973465, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1398349553346634, "sampling/sampling_logp_difference/max": 1.7621915340423584, "sampling/importance_sampling_ratio/min": 0.17166823148727417, "sampling/importance_sampling_ratio/mean": 1.0136052370071411, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8361148312687874, "clip_ratio/low_mean": 0.03289634082466364, "clip_ratio/low_min": 0.03289634082466364, "clip_ratio/high_mean": 0.08147118706256151, "clip_ratio/high_max": 0.08147118706256151, "clip_ratio/region_mean": 0.11436752788722515, "reward_total_mean": 0.5033431053161621, "reward_meter_mean": 0.3656251132488251, "reward_meter_std": 0.24074342846870422, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9848576784133911, "reward_repeat_soft_std": 0.006769011728465557, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19171687960624695, "reward_total_composite_mean": 0.5033431053161621, "reward_total_composite_std": 0.2377743273973465} {"timestamp_utc": "2026-04-13T03:44:09Z", "mode": "train", "global_step": 2158, "epoch": 0.2167754897036665, "loss": -0.0592, "grad_norm": 11.600316047668457, "learning_rate": 3.463636363636364e-06, "num_tokens": 3964639.0, "completions/mean_length": 61.0, "completions/min_length": 54.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.5675431489944458, "rewards/meter/std": 0.2381920963525772, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9378310441970825, "rewards/repeat_soft/std": 0.018750742077827454, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.6189274787902832, "rewards/total_composite/std": 0.09757263213396072, "reward": 0.6189274787902832, "reward_std": 0.09757263213396072, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09901092946529388, "sampling/sampling_logp_difference/max": 2.6352319717407227, "sampling/importance_sampling_ratio/min": 0.07170233875513077, "sampling/importance_sampling_ratio/mean": 1.0011636018753052, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4332125522196293, "clip_ratio/low_mean": 0.04372708545997739, "clip_ratio/low_min": 0.04372708545997739, "clip_ratio/high_mean": 0.03260448854416609, "clip_ratio/high_max": 0.03260448854416609, "clip_ratio/region_mean": 0.07633157400414348, "reward_total_mean": 0.6189274787902832, "reward_meter_mean": 0.5675431489944458, "reward_meter_std": 0.2381920963525772, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9378310441970825, "reward_repeat_soft_std": 0.018750742077827454, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.6189274787902832, "reward_total_composite_std": 0.09757263213396072} {"timestamp_utc": "2026-04-13T03:44:21Z", "mode": "train", "global_step": 2159, "epoch": 0.21687594173782018, "loss": -0.0805, "grad_norm": 5.117190361022949, "learning_rate": 3.460606060606061e-06, "num_tokens": 3966458.0, "completions/mean_length": 111.375, "completions/min_length": 49.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 54.142860412597656, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.7422869205474854, "rewards/meter/std": 0.44306397438049316, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9811512231826782, "rewards/repeat_soft/std": 0.015078697353601456, "rewards/judge_quality/mean": 0.5012499690055847, "rewards/judge_quality/std": 0.2796649634838104, "rewards/total_composite/mean": 0.7231442928314209, "rewards/total_composite/std": 0.2787795960903168, "reward": 0.7231442928314209, "reward_std": 0.278779536485672, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11574484407901764, "sampling/sampling_logp_difference/max": 2.0505056381225586, "sampling/importance_sampling_ratio/min": 0.1286698281764984, "sampling/importance_sampling_ratio/mean": 0.9981722235679626, "sampling/importance_sampling_ratio/max": 1.6945220232009888, "entropy": 0.591023750603199, "clip_ratio/low_mean": 0.012500000186264515, "clip_ratio/low_min": 0.012500000186264515, "clip_ratio/high_mean": 0.0888170781545341, "clip_ratio/high_max": 0.0888170781545341, "clip_ratio/region_mean": 0.10131707834079862, "reward_total_mean": 0.7231442928314209, "reward_meter_mean": 0.7422869205474854, "reward_meter_std": 0.44306397438049316, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9811512231826782, "reward_repeat_soft_std": 0.015078697353601456, "reward_judge_quality_mean": 0.5012499690055847, "reward_judge_quality_std": 0.2796649634838104, "reward_total_composite_mean": 0.7231442928314209, "reward_total_composite_std": 0.2787795960903168} {"timestamp_utc": "2026-04-13T03:44:29Z", "mode": "train", "global_step": 2160, "epoch": 0.2169763937719739, "loss": 0.0247, "grad_norm": 11.87191390991211, "learning_rate": 3.4575757575757577e-06, "num_tokens": 3968328.0, "completions/mean_length": 61.75, "completions/min_length": 56.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.75, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.7804895639419556, "rewards/meter/std": 0.261604368686676, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.985032320022583, "rewards/repeat_soft/std": 0.011082558892667294, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7130985260009766, "rewards/total_composite/std": 0.11066580563783646, "reward": 0.7130985260009766, "reward_std": 0.11066580563783646, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10055925697088242, "sampling/sampling_logp_difference/max": 1.251371145248413, "sampling/importance_sampling_ratio/min": 0.2861122190952301, "sampling/importance_sampling_ratio/mean": 1.0190556049346924, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5961615294218063, "clip_ratio/low_mean": 0.04717020969837904, "clip_ratio/low_min": 0.04717020969837904, "clip_ratio/high_mean": 0.06544131319969893, "clip_ratio/high_max": 0.06544131319969893, "clip_ratio/region_mean": 0.11261152289807796, "reward_total_mean": 0.7130985260009766, "reward_meter_mean": 0.7804895639419556, "reward_meter_std": 0.261604368686676, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.985032320022583, "reward_repeat_soft_std": 0.011082558892667294, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7130985260009766, "reward_total_composite_std": 0.11066580563783646} {"timestamp_utc": "2026-04-13T03:44:37Z", "mode": "train", "global_step": 2161, "epoch": 0.21707684580612757, "loss": 0.0161, "grad_norm": 8.839858055114746, "learning_rate": 3.454545454545455e-06, "num_tokens": 3970338.0, "completions/mean_length": 81.25, "completions/min_length": 76.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.25, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9460320472717285, "rewards/meter/std": 0.03207992762327194, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9807213544845581, "rewards/repeat_soft/std": 0.010018600150942802, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7870365381240845, "rewards/total_composite/std": 0.03421012684702873, "reward": 0.7870365381240845, "reward_std": 0.03421012684702873, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12747643887996674, "sampling/sampling_logp_difference/max": 2.1187081336975098, "sampling/importance_sampling_ratio/min": 0.16763858497142792, "sampling/importance_sampling_ratio/mean": 1.005447506904602, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.743398405611515, "clip_ratio/low_mean": 0.03203784953802824, "clip_ratio/low_min": 0.03203784953802824, "clip_ratio/high_mean": 0.08651833422482014, "clip_ratio/high_max": 0.08651833422482014, "clip_ratio/region_mean": 0.11855618376284838, "reward_total_mean": 0.7870365381240845, "reward_meter_mean": 0.9460320472717285, "reward_meter_std": 0.03207992762327194, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9807213544845581, "reward_repeat_soft_std": 0.010018600150942802, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7870365381240845, "reward_total_composite_std": 0.03421012684702873} {"timestamp_utc": "2026-04-13T03:44:48Z", "mode": "train", "global_step": 2162, "epoch": 0.21717729784028125, "loss": -0.1572, "grad_norm": 2.1029350757598877, "learning_rate": 3.451515151515152e-06, "num_tokens": 3972180.0, "completions/mean_length": 120.25, "completions/min_length": 59.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 64.28572082519531, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9074735641479492, "rewards/meter/std": 0.151740163564682, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9456265568733215, "rewards/repeat_soft/std": 0.04005679115653038, "rewards/judge_quality/mean": 0.543749988079071, "rewards/judge_quality/std": 0.2986128628253937, "rewards/total_composite/mean": 0.7276203036308289, "rewards/total_composite/std": 0.30334123969078064, "reward": 0.7276203036308289, "reward_std": 0.30334123969078064, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13320565223693848, "sampling/sampling_logp_difference/max": 1.500542163848877, "sampling/importance_sampling_ratio/min": 0.22300922870635986, "sampling/importance_sampling_ratio/mean": 1.0034997463226318, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.545332059264183, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.109089195728302, "clip_ratio/high_max": 0.109089195728302, "clip_ratio/region_mean": 0.109089195728302, "reward_total_mean": 0.7276203036308289, "reward_meter_mean": 0.9074735641479492, "reward_meter_std": 0.151740163564682, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9456265568733215, "reward_repeat_soft_std": 0.04005679115653038, "reward_judge_quality_mean": 0.543749988079071, "reward_judge_quality_std": 0.2986128628253937, "reward_total_composite_mean": 0.7276203036308289, "reward_total_composite_std": 0.30334123969078064} {"timestamp_utc": "2026-04-13T03:44:55Z", "mode": "train", "global_step": 2163, "epoch": 0.21727774987443496, "loss": 0.0227, "grad_norm": 10.450579643249512, "learning_rate": 3.4484848484848486e-06, "num_tokens": 3973914.0, "completions/mean_length": 48.75, "completions/min_length": 45.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.75, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.7772257328033447, "rewards/meter/std": 0.34275323152542114, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.926811933517456, "rewards/repeat_soft/std": 0.05441522225737572, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.713182806968689, "rewards/total_composite/std": 0.1519807130098343, "reward": 0.713182806968689, "reward_std": 0.1519807130098343, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11457902193069458, "sampling/sampling_logp_difference/max": 2.4055652618408203, "sampling/importance_sampling_ratio/min": 0.09021448343992233, "sampling/importance_sampling_ratio/mean": 1.017221450805664, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6004196181893349, "clip_ratio/low_mean": 0.032870370894670486, "clip_ratio/low_min": 0.032870370894670486, "clip_ratio/high_mean": 0.07280077785253525, "clip_ratio/high_max": 0.07280077785253525, "clip_ratio/region_mean": 0.10567114874720573, "reward_total_mean": 0.713182806968689, "reward_meter_mean": 0.7772257328033447, "reward_meter_std": 0.34275323152542114, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.926811933517456, "reward_repeat_soft_std": 0.05441522225737572, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.713182806968689, "reward_total_composite_std": 0.1519807130098343} {"timestamp_utc": "2026-04-13T03:45:02Z", "mode": "train", "global_step": 2164, "epoch": 0.21737820190858864, "loss": 0.0151, "grad_norm": 12.086810111999512, "learning_rate": 3.445454545454546e-06, "num_tokens": 3975715.0, "completions/mean_length": 61.125, "completions/min_length": 53.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.125, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.8463197946548462, "rewards/meter/std": 0.28287047147750854, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9962947368621826, "rewards/repeat_soft/std": 0.0035038681235164404, "rewards/judge_quality/mean": 0.5399999618530273, "rewards/judge_quality/std": 0.14957083761692047, "rewards/total_composite/mean": 0.792473316192627, "rewards/total_composite/std": 0.10456790030002594, "reward": 0.792473316192627, "reward_std": 0.10456789284944534, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13827243447303772, "sampling/sampling_logp_difference/max": 3.10457706451416, "sampling/importance_sampling_ratio/min": 0.04484347999095917, "sampling/importance_sampling_ratio/mean": 1.0052381753921509, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7129215970635414, "clip_ratio/low_mean": 0.029338153079152107, "clip_ratio/low_min": 0.029338153079152107, "clip_ratio/high_mean": 0.08932743035256863, "clip_ratio/high_max": 0.08932743035256863, "clip_ratio/region_mean": 0.11866558343172073, "reward_total_mean": 0.792473316192627, "reward_meter_mean": 0.8463197946548462, "reward_meter_std": 0.28287047147750854, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9962947368621826, "reward_repeat_soft_std": 0.0035038681235164404, "reward_judge_quality_mean": 0.5399999618530273, "reward_judge_quality_std": 0.14957083761692047, "reward_total_composite_mean": 0.792473316192627, "reward_total_composite_std": 0.10456790030002594} {"timestamp_utc": "2026-04-13T03:45:08Z", "mode": "train", "global_step": 2165, "epoch": 0.21747865394274235, "loss": 0.0162, "grad_norm": 14.090245246887207, "learning_rate": 3.4424242424242427e-06, "num_tokens": 3977349.0, "completions/mean_length": 33.25, "completions/min_length": 30.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.25, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.8437027931213379, "rewards/meter/std": 0.3339202404022217, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9604166746139526, "rewards/repeat_soft/std": 0.00589255103841424, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7573329210281372, "rewards/total_composite/std": 0.14922836422920227, "reward": 0.7573329210281372, "reward_std": 0.14922837913036346, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12803924083709717, "sampling/sampling_logp_difference/max": 1.5484447479248047, "sampling/importance_sampling_ratio/min": 0.21257832646369934, "sampling/importance_sampling_ratio/mean": 1.005717396736145, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8770985305309296, "clip_ratio/low_mean": 0.02890625037252903, "clip_ratio/low_min": 0.02890625037252903, "clip_ratio/high_mean": 0.10377998929470778, "clip_ratio/high_max": 0.10377998929470778, "clip_ratio/region_mean": 0.1326862396672368, "reward_total_mean": 0.7573329210281372, "reward_meter_mean": 0.8437027931213379, "reward_meter_std": 0.3339202404022217, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9604166746139526, "reward_repeat_soft_std": 0.00589255103841424, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7573329210281372, "reward_total_composite_std": 0.14922836422920227} {"timestamp_utc": "2026-04-13T03:45:20Z", "mode": "train", "global_step": 2166, "epoch": 0.21757910597689603, "loss": -0.1483, "grad_norm": 4.048954486846924, "learning_rate": 3.4393939393939395e-06, "num_tokens": 3980141.0, "completions/mean_length": 208.0, "completions/min_length": 143.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 164.57144165039062, "completions/min_terminated_length": 143.0, "completions/max_terminated_length": 178.0, "rewards/meter/mean": 0.9143844246864319, "rewards/meter/std": 0.2008446753025055, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9737042188644409, "rewards/repeat_soft/std": 0.006069052033126354, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.6125805974006653, "rewards/total_composite/std": 0.378096342086792, "reward": 0.6125805974006653, "reward_std": 0.378096342086792, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12370274215936661, "sampling/sampling_logp_difference/max": 2.809138059616089, "sampling/importance_sampling_ratio/min": 0.06025690957903862, "sampling/importance_sampling_ratio/mean": 1.005443811416626, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7090008780360222, "clip_ratio/low_mean": 0.013888888992369175, "clip_ratio/low_min": 0.013888888992369175, "clip_ratio/high_mean": 0.09888195339590311, "clip_ratio/high_max": 0.09888195339590311, "clip_ratio/region_mean": 0.11277084238827229, "reward_total_mean": 0.6125805974006653, "reward_meter_mean": 0.9143844246864319, "reward_meter_std": 0.2008446753025055, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9737042188644409, "reward_repeat_soft_std": 0.006069052033126354, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.6125805974006653, "reward_total_composite_std": 0.378096342086792} {"timestamp_utc": "2026-04-13T03:45:27Z", "mode": "train", "global_step": 2167, "epoch": 0.2176795580110497, "loss": 0.0469, "grad_norm": 10.098995208740234, "learning_rate": 3.4363636363636364e-06, "num_tokens": 3981914.0, "completions/mean_length": 53.625, "completions/min_length": 51.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.625, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.83023601770401, "rewards/meter/std": 0.2868667542934418, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9909830689430237, "rewards/repeat_soft/std": 0.023015351966023445, "rewards/judge_quality/mean": 0.5900000333786011, "rewards/judge_quality/std": 0.22696760296821594, "rewards/total_composite/mean": 0.7997045516967773, "rewards/total_composite/std": 0.12125399708747864, "reward": 0.7997045516967773, "reward_std": 0.12125399708747864, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10319732874631882, "sampling/sampling_logp_difference/max": 2.562135934829712, "sampling/importance_sampling_ratio/min": 0.07713979482650757, "sampling/importance_sampling_ratio/mean": 0.9874960780143738, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5381019078195095, "clip_ratio/low_mean": 0.02494553430005908, "clip_ratio/low_min": 0.02494553430005908, "clip_ratio/high_mean": 0.06407301966100931, "clip_ratio/high_max": 0.06407301966100931, "clip_ratio/region_mean": 0.08901855396106839, "reward_total_mean": 0.7997045516967773, "reward_meter_mean": 0.83023601770401, "reward_meter_std": 0.2868667542934418, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9909830689430237, "reward_repeat_soft_std": 0.023015351966023445, "reward_judge_quality_mean": 0.5900000333786011, "reward_judge_quality_std": 0.22696760296821594, "reward_total_composite_mean": 0.7997045516967773, "reward_total_composite_std": 0.12125399708747864} {"timestamp_utc": "2026-04-13T03:45:34Z", "mode": "train", "global_step": 2168, "epoch": 0.21778001004520342, "loss": -0.0028, "grad_norm": 13.568170547485352, "learning_rate": 3.4333333333333336e-06, "num_tokens": 3983606.0, "completions/mean_length": 49.5, "completions/min_length": 41.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.5, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.8558423519134521, "rewards/meter/std": 0.29309746623039246, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9770315885543823, "rewards/repeat_soft/std": 0.024760030210018158, "rewards/judge_quality/mean": 0.5249999761581421, "rewards/judge_quality/std": 0.13887302577495575, "rewards/total_composite/mean": 0.7903321981430054, "rewards/total_composite/std": 0.1479955017566681, "reward": 0.7903321981430054, "reward_std": 0.1479954868555069, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09935560077428818, "sampling/sampling_logp_difference/max": 1.7210557460784912, "sampling/importance_sampling_ratio/min": 0.29452452063560486, "sampling/importance_sampling_ratio/mean": 1.00705885887146, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6473883576691151, "clip_ratio/low_mean": 0.03509208559989929, "clip_ratio/low_min": 0.03509208559989929, "clip_ratio/high_mean": 0.11515654157847166, "clip_ratio/high_max": 0.11515654157847166, "clip_ratio/region_mean": 0.15024862717837095, "reward_total_mean": 0.7903321981430054, "reward_meter_mean": 0.8558423519134521, "reward_meter_std": 0.29309746623039246, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9770315885543823, "reward_repeat_soft_std": 0.024760030210018158, "reward_judge_quality_mean": 0.5249999761581421, "reward_judge_quality_std": 0.13887302577495575, "reward_total_composite_mean": 0.7903321981430054, "reward_total_composite_std": 0.1479955017566681} {"timestamp_utc": "2026-04-13T03:45:40Z", "mode": "train", "global_step": 2169, "epoch": 0.2178804620793571, "loss": 0.0032, "grad_norm": 19.802692413330078, "learning_rate": 3.4303030303030305e-06, "num_tokens": 3985143.0, "completions/mean_length": 26.125, "completions/min_length": 21.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.125, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9828475117683411, "rewards/meter/std": 0.015862274914979935, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8167813420295715, "rewards/total_composite/std": 0.006177342962473631, "reward": 0.8167813420295715, "reward_std": 0.006177341565489769, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1036790981888771, "sampling/sampling_logp_difference/max": 1.2104125022888184, "sampling/importance_sampling_ratio/min": 0.29807430505752563, "sampling/importance_sampling_ratio/mean": 1.0379753112792969, "sampling/importance_sampling_ratio/max": 1.9737977981567383, "entropy": 0.5779238045215607, "clip_ratio/low_mean": 0.011982570867985487, "clip_ratio/low_min": 0.011982570867985487, "clip_ratio/high_mean": 0.050158515106886625, "clip_ratio/high_max": 0.050158515106886625, "clip_ratio/region_mean": 0.06214108597487211, "reward_total_mean": 0.8167813420295715, "reward_meter_mean": 0.9828475117683411, "reward_meter_std": 0.015862274914979935, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8167813420295715, "reward_total_composite_std": 0.006177342962473631} {"timestamp_utc": "2026-04-13T03:45:52Z", "mode": "train", "global_step": 2170, "epoch": 0.2179809141135108, "loss": -0.1788, "grad_norm": 2.445491075515747, "learning_rate": 3.4272727272727273e-06, "num_tokens": 3987126.0, "completions/mean_length": 148.875, "completions/min_length": 82.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 97.00000762939453, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.5344013571739197, "rewards/meter/std": 0.2629765570163727, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9766893982887268, "rewards/repeat_soft/std": 0.030244482681155205, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.18845234811306, "rewards/total_composite/mean": 0.5673462152481079, "rewards/total_composite/std": 0.2409856617450714, "reward": 0.5673462152481079, "reward_std": 0.2409856617450714, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12876369059085846, "sampling/sampling_logp_difference/max": 1.78403902053833, "sampling/importance_sampling_ratio/min": 0.16795839369297028, "sampling/importance_sampling_ratio/mean": 0.9932206869125366, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5752340778708458, "clip_ratio/low_mean": 0.023298109881579876, "clip_ratio/low_min": 0.023298109881579876, "clip_ratio/high_mean": 0.07734532095491886, "clip_ratio/high_max": 0.07734532095491886, "clip_ratio/region_mean": 0.10064343083649874, "reward_total_mean": 0.5673462152481079, "reward_meter_mean": 0.5344013571739197, "reward_meter_std": 0.2629765570163727, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9766893982887268, "reward_repeat_soft_std": 0.030244482681155205, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.18845234811306, "reward_total_composite_mean": 0.5673462152481079, "reward_total_composite_std": 0.2409856617450714} {"timestamp_utc": "2026-04-13T03:45:58Z", "mode": "train", "global_step": 2171, "epoch": 0.2180813661476645, "loss": -0.0616, "grad_norm": 15.564566612243652, "learning_rate": 3.4242424242424246e-06, "num_tokens": 3988664.0, "completions/mean_length": 34.25, "completions/min_length": 30.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.25, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.991759181022644, "rewards/meter/std": 0.007669894024729729, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9594236612319946, "rewards/repeat_soft/std": 0.005712059326469898, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.09953463077545166, "rewards/total_composite/mean": 0.8249840140342712, "rewards/total_composite/std": 0.030548451468348503, "reward": 0.8249840140342712, "reward_std": 0.03054845705628395, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1236998587846756, "sampling/sampling_logp_difference/max": 1.867807388305664, "sampling/importance_sampling_ratio/min": 0.15446196496486664, "sampling/importance_sampling_ratio/mean": 0.986853837966919, "sampling/importance_sampling_ratio/max": 1.7770904302597046, "entropy": 0.6995250955224037, "clip_ratio/low_mean": 0.03820695448666811, "clip_ratio/low_min": 0.03820695448666811, "clip_ratio/high_mean": 0.07304832432419062, "clip_ratio/high_max": 0.07304832432419062, "clip_ratio/region_mean": 0.11125527881085873, "reward_total_mean": 0.8249840140342712, "reward_meter_mean": 0.991759181022644, "reward_meter_std": 0.007669894024729729, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9594236612319946, "reward_repeat_soft_std": 0.005712059326469898, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.09953463077545166, "reward_total_composite_mean": 0.8249840140342712, "reward_total_composite_std": 0.030548451468348503} {"timestamp_utc": "2026-04-13T03:46:06Z", "mode": "train", "global_step": 2172, "epoch": 0.21818181818181817, "loss": 0.0655, "grad_norm": 11.130759239196777, "learning_rate": 3.4212121212121214e-06, "num_tokens": 3990403.0, "completions/mean_length": 62.375, "completions/min_length": 53.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.375, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.7718318700790405, "rewards/meter/std": 0.32924649119377136, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9917565584182739, "rewards/repeat_soft/std": 0.010434611700475216, "rewards/judge_quality/mean": 0.42499998211860657, "rewards/judge_quality/std": 0.0707106739282608, "rewards/total_composite/mean": 0.7239999771118164, "rewards/total_composite/std": 0.16019387543201447, "reward": 0.7239999771118164, "reward_std": 0.16019389033317566, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13216406106948853, "sampling/sampling_logp_difference/max": 2.0273404121398926, "sampling/importance_sampling_ratio/min": 0.1316852867603302, "sampling/importance_sampling_ratio/mean": 1.0245612859725952, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.89425028860569, "clip_ratio/low_mean": 0.02469499222934246, "clip_ratio/low_min": 0.02469499222934246, "clip_ratio/high_mean": 0.08926363382488489, "clip_ratio/high_max": 0.08926363382488489, "clip_ratio/region_mean": 0.11395862605422735, "reward_total_mean": 0.7239999771118164, "reward_meter_mean": 0.7718318700790405, "reward_meter_std": 0.32924649119377136, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9917565584182739, "reward_repeat_soft_std": 0.010434611700475216, "reward_judge_quality_mean": 0.42499998211860657, "reward_judge_quality_std": 0.0707106739282608, "reward_total_composite_mean": 0.7239999771118164, "reward_total_composite_std": 0.16019387543201447} {"timestamp_utc": "2026-04-13T03:46:14Z", "mode": "train", "global_step": 2173, "epoch": 0.21828227021597188, "loss": 0.0568, "grad_norm": 10.413093566894531, "learning_rate": 3.4181818181818182e-06, "num_tokens": 3992151.0, "completions/mean_length": 62.5, "completions/min_length": 58.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.5, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.7844523191452026, "rewards/meter/std": 0.3823116719722748, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.98427814245224, "rewards/repeat_soft/std": 0.020582180470228195, "rewards/judge_quality/mean": 0.543749988079071, "rewards/judge_quality/std": 0.22038522362709045, "rewards/total_composite/mean": 0.764556348323822, "rewards/total_composite/std": 0.17127083241939545, "reward": 0.764556348323822, "reward_std": 0.17127083241939545, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12476146221160889, "sampling/sampling_logp_difference/max": 2.2380928993225098, "sampling/importance_sampling_ratio/min": 0.10666172206401825, "sampling/importance_sampling_ratio/mean": 0.9909247159957886, "sampling/importance_sampling_ratio/max": 1.6395636796951294, "entropy": 0.6514529809355736, "clip_ratio/low_mean": 0.03456439450383186, "clip_ratio/low_min": 0.03456439450383186, "clip_ratio/high_mean": 0.07885161880403757, "clip_ratio/high_max": 0.07885161880403757, "clip_ratio/region_mean": 0.11341601330786943, "reward_total_mean": 0.764556348323822, "reward_meter_mean": 0.7844523191452026, "reward_meter_std": 0.3823116719722748, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.98427814245224, "reward_repeat_soft_std": 0.020582180470228195, "reward_judge_quality_mean": 0.543749988079071, "reward_judge_quality_std": 0.22038522362709045, "reward_total_composite_mean": 0.764556348323822, "reward_total_composite_std": 0.17127083241939545} {"timestamp_utc": "2026-04-13T03:46:23Z", "mode": "train", "global_step": 2174, "epoch": 0.21838272225012556, "loss": -0.0087, "grad_norm": 12.25987720489502, "learning_rate": 3.415151515151515e-06, "num_tokens": 3993827.0, "completions/mean_length": 57.5, "completions/min_length": 52.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.5, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.8736085295677185, "rewards/meter/std": 0.22467653453350067, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.986661970615387, "rewards/repeat_soft/std": 0.01321916189044714, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.10260014235973358, "rewards/total_composite/mean": 0.7824150323867798, "rewards/total_composite/std": 0.09456533193588257, "reward": 0.7824150323867798, "reward_std": 0.09456531703472137, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11870544403791428, "sampling/sampling_logp_difference/max": 1.20670747756958, "sampling/importance_sampling_ratio/min": 0.2991807162761688, "sampling/importance_sampling_ratio/mean": 0.991826593875885, "sampling/importance_sampling_ratio/max": 1.8289244174957275, "entropy": 0.6922493949532509, "clip_ratio/low_mean": 0.0024038462433964014, "clip_ratio/low_min": 0.0024038462433964014, "clip_ratio/high_mean": 0.10590283526107669, "clip_ratio/high_max": 0.10590283526107669, "clip_ratio/region_mean": 0.10830668150447309, "reward_total_mean": 0.7824150323867798, "reward_meter_mean": 0.8736085295677185, "reward_meter_std": 0.22467653453350067, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.986661970615387, "reward_repeat_soft_std": 0.01321916189044714, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.10260014235973358, "reward_total_composite_mean": 0.7824150323867798, "reward_total_composite_std": 0.09456533193588257} {"timestamp_utc": "2026-04-13T03:46:31Z", "mode": "train", "global_step": 2175, "epoch": 0.21848317428427927, "loss": -0.0534, "grad_norm": 10.612059593200684, "learning_rate": 3.4121212121212123e-06, "num_tokens": 3995567.0, "completions/mean_length": 50.5, "completions/min_length": 43.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.5, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9131566882133484, "rewards/meter/std": 0.13473257422447205, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.992901623249054, "rewards/repeat_soft/std": 0.005452314391732216, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.0975411981344223, "rewards/total_composite/mean": 0.7772107124328613, "rewards/total_composite/std": 0.06168902665376663, "reward": 0.7772107124328613, "reward_std": 0.061689022928476334, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13773812353610992, "sampling/sampling_logp_difference/max": 2.2045087814331055, "sampling/importance_sampling_ratio/min": 0.11030469089746475, "sampling/importance_sampling_ratio/mean": 1.03038489818573, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7950246706604958, "clip_ratio/low_mean": 0.05445736646652222, "clip_ratio/low_min": 0.05445736646652222, "clip_ratio/high_mean": 0.08349244575947523, "clip_ratio/high_max": 0.08349244575947523, "clip_ratio/region_mean": 0.13794981222599745, "reward_total_mean": 0.7772107124328613, "reward_meter_mean": 0.9131566882133484, "reward_meter_std": 0.13473257422447205, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.992901623249054, "reward_repeat_soft_std": 0.005452314391732216, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.0975411981344223, "reward_total_composite_mean": 0.7772107124328613, "reward_total_composite_std": 0.06168902665376663} {"timestamp_utc": "2026-04-13T03:46:38Z", "mode": "train", "global_step": 2176, "epoch": 0.21858362631843295, "loss": 0.0205, "grad_norm": 13.514725685119629, "learning_rate": 3.409090909090909e-06, "num_tokens": 3997087.0, "completions/mean_length": 46.0, "completions/min_length": 42.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.0, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.986920952796936, "rewards/meter/std": 0.009752495214343071, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9326457977294922, "rewards/repeat_soft/std": 0.04510067030787468, "rewards/judge_quality/mean": 0.36374998092651367, "rewards/judge_quality/std": 0.09500939399003983, "rewards/total_composite/mean": 0.796504020690918, "rewards/total_composite/std": 0.030632728710770607, "reward": 0.796504020690918, "reward_std": 0.03063272126019001, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14959143102169037, "sampling/sampling_logp_difference/max": 3.864570140838623, "sampling/importance_sampling_ratio/min": 0.020971937105059624, "sampling/importance_sampling_ratio/mean": 1.0158486366271973, "sampling/importance_sampling_ratio/max": 1.9475743770599365, "entropy": 0.8575422614812851, "clip_ratio/low_mean": 0.03638881538063288, "clip_ratio/low_min": 0.03638881538063288, "clip_ratio/high_mean": 0.09394605550915003, "clip_ratio/high_max": 0.09394605550915003, "clip_ratio/region_mean": 0.1303348708897829, "reward_total_mean": 0.796504020690918, "reward_meter_mean": 0.986920952796936, "reward_meter_std": 0.009752495214343071, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9326457977294922, "reward_repeat_soft_std": 0.04510067030787468, "reward_judge_quality_mean": 0.36374998092651367, "reward_judge_quality_std": 0.09500939399003983, "reward_total_composite_mean": 0.796504020690918, "reward_total_composite_std": 0.030632728710770607} {"timestamp_utc": "2026-04-13T03:46:46Z", "mode": "train", "global_step": 2177, "epoch": 0.21868407835258663, "loss": 0.0391, "grad_norm": 8.716300010681152, "learning_rate": 3.406060606060606e-06, "num_tokens": 3998963.0, "completions/mean_length": 64.5, "completions/min_length": 56.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.5, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9864214062690735, "rewards/meter/std": 0.008640571497380733, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9522553086280823, "rewards/repeat_soft/std": 0.05240863561630249, "rewards/judge_quality/mean": 0.5475000143051147, "rewards/judge_quality/std": 0.15191635489463806, "rewards/total_composite/mean": 0.8533651828765869, "rewards/total_composite/std": 0.04949261620640755, "reward": 0.8533651828765869, "reward_std": 0.04949261248111725, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10749881714582443, "sampling/sampling_logp_difference/max": 2.00602650642395, "sampling/importance_sampling_ratio/min": 0.13452214002609253, "sampling/importance_sampling_ratio/mean": 1.0314910411834717, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6140121445059776, "clip_ratio/low_mean": 0.059205726720392704, "clip_ratio/low_min": 0.059205726720392704, "clip_ratio/high_mean": 0.05028488766402006, "clip_ratio/high_max": 0.05028488766402006, "clip_ratio/region_mean": 0.10949061438441277, "reward_total_mean": 0.8533651828765869, "reward_meter_mean": 0.9864214062690735, "reward_meter_std": 0.008640571497380733, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9522553086280823, "reward_repeat_soft_std": 0.05240863561630249, "reward_judge_quality_mean": 0.5475000143051147, "reward_judge_quality_std": 0.15191635489463806, "reward_total_composite_mean": 0.8533651828765869, "reward_total_composite_std": 0.04949261620640755} {"timestamp_utc": "2026-04-13T03:46:59Z", "mode": "train", "global_step": 2178, "epoch": 0.21878453038674034, "loss": -0.1943, "grad_norm": 1.8049218654632568, "learning_rate": 3.4030303030303036e-06, "num_tokens": 4001091.0, "completions/mean_length": 145.0, "completions/min_length": 89.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 92.5714340209961, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.8741880655288696, "rewards/meter/std": 0.32743769884109497, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8757907152175903, "rewards/repeat_soft/std": 0.08044612407684326, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.7000243663787842, "rewards/total_composite/std": 0.28342366218566895, "reward": 0.7000243663787842, "reward_std": 0.28342366218566895, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10044372081756592, "sampling/sampling_logp_difference/max": 2.5648674964904785, "sampling/importance_sampling_ratio/min": 0.07692937552928925, "sampling/importance_sampling_ratio/mean": 1.0104094743728638, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.46456580236554146, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0703620738349855, "clip_ratio/high_max": 0.0703620738349855, "clip_ratio/region_mean": 0.0703620738349855, "reward_total_mean": 0.7000243663787842, "reward_meter_mean": 0.8741880655288696, "reward_meter_std": 0.32743769884109497, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8757907152175903, "reward_repeat_soft_std": 0.08044612407684326, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.7000243663787842, "reward_total_composite_std": 0.28342366218566895} {"timestamp_utc": "2026-04-13T03:47:13Z", "mode": "train", "global_step": 2179, "epoch": 0.21888498242089402, "loss": -0.0736, "grad_norm": 3.5441646575927734, "learning_rate": 3.4000000000000005e-06, "num_tokens": 4002495.0, "completions/mean_length": 88.5, "completions/min_length": 23.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 28.000001907348633, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.6003206968307495, "rewards/meter/std": 0.4232023060321808, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9517265558242798, "rewards/repeat_soft/std": 0.020055795088410378, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.23445606231689453, "rewards/total_composite/mean": 0.5992839336395264, "rewards/total_composite/std": 0.316037654876709, "reward": 0.5992839336395264, "reward_std": 0.316037654876709, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10498762875795364, "sampling/sampling_logp_difference/max": 1.3628560304641724, "sampling/importance_sampling_ratio/min": 0.25592878460884094, "sampling/importance_sampling_ratio/mean": 1.0192292928695679, "sampling/importance_sampling_ratio/max": 1.8733153343200684, "entropy": 0.5060436353087425, "clip_ratio/low_mean": 0.036395430099219084, "clip_ratio/low_min": 0.036395430099219084, "clip_ratio/high_mean": 0.04861772712320089, "clip_ratio/high_max": 0.04861772712320089, "clip_ratio/region_mean": 0.08501315722241998, "reward_total_mean": 0.5992839336395264, "reward_meter_mean": 0.6003206968307495, "reward_meter_std": 0.4232023060321808, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9517265558242798, "reward_repeat_soft_std": 0.020055795088410378, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.23445606231689453, "reward_total_composite_mean": 0.5992839336395264, "reward_total_composite_std": 0.316037654876709} {"timestamp_utc": "2026-04-13T03:47:21Z", "mode": "train", "global_step": 2180, "epoch": 0.21898543445504773, "loss": 0.0282, "grad_norm": 10.374370574951172, "learning_rate": 3.3969696969696973e-06, "num_tokens": 4004457.0, "completions/mean_length": 79.25, "completions/min_length": 73.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.25, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.9723964929580688, "rewards/meter/std": 0.02354084886610508, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9703537225723267, "rewards/repeat_soft/std": 0.01773662120103836, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.8229888081550598, "rewards/total_composite/std": 0.053443793207407, "reward": 0.8229888081550598, "reward_std": 0.053443778306245804, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12893055379390717, "sampling/sampling_logp_difference/max": 1.61069917678833, "sampling/importance_sampling_ratio/min": 0.19974790513515472, "sampling/importance_sampling_ratio/mean": 1.0167304277420044, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8020729422569275, "clip_ratio/low_mean": 0.09860356338322163, "clip_ratio/low_min": 0.09860356338322163, "clip_ratio/high_mean": 0.012820512987673283, "clip_ratio/high_max": 0.012820512987673283, "clip_ratio/region_mean": 0.11142407637089491, "reward_total_mean": 0.8229888081550598, "reward_meter_mean": 0.9723964929580688, "reward_meter_std": 0.02354084886610508, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9703537225723267, "reward_repeat_soft_std": 0.01773662120103836, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.8229888081550598, "reward_total_composite_std": 0.053443793207407} {"timestamp_utc": "2026-04-13T03:47:29Z", "mode": "train", "global_step": 2181, "epoch": 0.2190858864892014, "loss": 0.0053, "grad_norm": 6.498655796051025, "learning_rate": 3.3939393939393946e-06, "num_tokens": 4006676.0, "completions/mean_length": 110.375, "completions/min_length": 104.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.375, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.9664720296859741, "rewards/meter/std": 0.030842741951346397, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9642898440361023, "rewards/repeat_soft/std": 0.01794886216521263, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7945914268493652, "rewards/total_composite/std": 0.029270615428686142, "reward": 0.7945914268493652, "reward_std": 0.029270613566040993, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10463643819093704, "sampling/sampling_logp_difference/max": 3.0485997200012207, "sampling/importance_sampling_ratio/min": 0.0474252887070179, "sampling/importance_sampling_ratio/mean": 1.0094376802444458, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5471298657357693, "clip_ratio/low_mean": 0.03679252043366432, "clip_ratio/low_min": 0.03679252043366432, "clip_ratio/high_mean": 0.06500415969640017, "clip_ratio/high_max": 0.06500415969640017, "clip_ratio/region_mean": 0.10179668013006449, "reward_total_mean": 0.7945914268493652, "reward_meter_mean": 0.9664720296859741, "reward_meter_std": 0.030842741951346397, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9642898440361023, "reward_repeat_soft_std": 0.01794886216521263, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7945914268493652, "reward_total_composite_std": 0.029270615428686142} {"timestamp_utc": "2026-04-13T03:47:38Z", "mode": "train", "global_step": 2182, "epoch": 0.2191863385233551, "loss": 0.0246, "grad_norm": 6.859695911407471, "learning_rate": 3.3909090909090914e-06, "num_tokens": 4009088.0, "completions/mean_length": 110.5, "completions/min_length": 107.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.5, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.9873310327529907, "rewards/meter/std": 0.0026621795259416103, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8765081167221069, "rewards/repeat_soft/std": 0.07585444301366806, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.819199800491333, "rewards/total_composite/std": 0.03217873349785805, "reward": 0.819199800491333, "reward_std": 0.032178737223148346, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11004900932312012, "sampling/sampling_logp_difference/max": 4.9084882736206055, "sampling/importance_sampling_ratio/min": 0.007383641786873341, "sampling/importance_sampling_ratio/mean": 1.0027936697006226, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5391822755336761, "clip_ratio/low_mean": 0.06620031781494617, "clip_ratio/low_min": 0.06620031781494617, "clip_ratio/high_mean": 0.01285046711564064, "clip_ratio/high_max": 0.01285046711564064, "clip_ratio/region_mean": 0.07905078493058681, "reward_total_mean": 0.819199800491333, "reward_meter_mean": 0.9873310327529907, "reward_meter_std": 0.0026621795259416103, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8765081167221069, "reward_repeat_soft_std": 0.07585444301366806, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.819199800491333, "reward_total_composite_std": 0.03217873349785805} {"timestamp_utc": "2026-04-13T03:47:50Z", "mode": "train", "global_step": 2183, "epoch": 0.2192867905575088, "loss": -0.1247, "grad_norm": 2.8295485973358154, "learning_rate": 3.3878787878787882e-06, "num_tokens": 4010904.0, "completions/mean_length": 118.0, "completions/min_length": 57.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 61.71428680419922, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.5565860271453857, "rewards/meter/std": 0.47139209508895874, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9617083072662354, "rewards/repeat_soft/std": 0.05450617894530296, "rewards/judge_quality/mean": 0.39625000953674316, "rewards/judge_quality/std": 0.14029940962791443, "rewards/total_composite/mean": 0.5823845267295837, "rewards/total_composite/std": 0.2981272041797638, "reward": 0.5823845267295837, "reward_std": 0.2981272041797638, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11485438793897629, "sampling/sampling_logp_difference/max": 1.288034439086914, "sampling/importance_sampling_ratio/min": 0.27581238746643066, "sampling/importance_sampling_ratio/mean": 1.0068411827087402, "sampling/importance_sampling_ratio/max": 1.8838509321212769, "entropy": 0.5513783097267151, "clip_ratio/low_mean": 0.038539353758096695, "clip_ratio/low_min": 0.038539353758096695, "clip_ratio/high_mean": 0.06571911182254553, "clip_ratio/high_max": 0.06571911182254553, "clip_ratio/region_mean": 0.10425846558064222, "reward_total_mean": 0.5823845267295837, "reward_meter_mean": 0.5565860271453857, "reward_meter_std": 0.47139209508895874, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9617083072662354, "reward_repeat_soft_std": 0.05450617894530296, "reward_judge_quality_mean": 0.39625000953674316, "reward_judge_quality_std": 0.14029940962791443, "reward_total_composite_mean": 0.5823845267295837, "reward_total_composite_std": 0.2981272041797638} {"timestamp_utc": "2026-04-13T03:48:01Z", "mode": "train", "global_step": 2184, "epoch": 0.21938724259166248, "loss": -0.1454, "grad_norm": 2.2945988178253174, "learning_rate": 3.384848484848485e-06, "num_tokens": 4012491.0, "completions/mean_length": 115.375, "completions/min_length": 54.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 58.71428680419922, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.7942250967025757, "rewards/meter/std": 0.26324018836021423, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9877901077270508, "rewards/repeat_soft/std": 0.0186786986887455, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.6641218662261963, "rewards/total_composite/std": 0.2830196022987366, "reward": 0.6641218662261963, "reward_std": 0.2830195724964142, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1248113214969635, "sampling/sampling_logp_difference/max": 1.5921664237976074, "sampling/importance_sampling_ratio/min": 0.20348429679870605, "sampling/importance_sampling_ratio/mean": 1.0015051364898682, "sampling/importance_sampling_ratio/max": 1.7318778038024902, "entropy": 0.617277130484581, "clip_ratio/low_mean": 0.02612673956900835, "clip_ratio/low_min": 0.02612673956900835, "clip_ratio/high_mean": 0.095067597925663, "clip_ratio/high_max": 0.095067597925663, "clip_ratio/region_mean": 0.12119433749467134, "reward_total_mean": 0.6641218662261963, "reward_meter_mean": 0.7942250967025757, "reward_meter_std": 0.26324018836021423, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9877901077270508, "reward_repeat_soft_std": 0.0186786986887455, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.6641218662261963, "reward_total_composite_std": 0.2830196022987366} {"timestamp_utc": "2026-04-13T03:48:07Z", "mode": "train", "global_step": 2185, "epoch": 0.21948769462581616, "loss": 0.0798, "grad_norm": 12.516645431518555, "learning_rate": 3.3818181818181823e-06, "num_tokens": 4013925.0, "completions/mean_length": 31.25, "completions/min_length": 28.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.25, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.911372184753418, "rewards/meter/std": 0.17524999380111694, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9585617184638977, "rewards/repeat_soft/std": 0.005944323260337114, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.18873640894889832, "rewards/total_composite/mean": 0.8142236471176147, "rewards/total_composite/std": 0.11007038503885269, "reward": 0.8142236471176147, "reward_std": 0.11007039248943329, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10667774081230164, "sampling/sampling_logp_difference/max": 1.5898613929748535, "sampling/importance_sampling_ratio/min": 0.20395387709140778, "sampling/importance_sampling_ratio/mean": 1.0069398880004883, "sampling/importance_sampling_ratio/max": 1.7121330499649048, "entropy": 0.7158937528729439, "clip_ratio/low_mean": 0.06460973434150219, "clip_ratio/low_min": 0.06460973434150219, "clip_ratio/high_mean": 0.06058509834110737, "clip_ratio/high_max": 0.06058509834110737, "clip_ratio/region_mean": 0.12519483268260956, "reward_total_mean": 0.8142236471176147, "reward_meter_mean": 0.911372184753418, "reward_meter_std": 0.17524999380111694, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9585617184638977, "reward_repeat_soft_std": 0.005944323260337114, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.18873640894889832, "reward_total_composite_mean": 0.8142236471176147, "reward_total_composite_std": 0.11007038503885269} {"timestamp_utc": "2026-04-13T03:48:14Z", "mode": "train", "global_step": 2186, "epoch": 0.21958814665996987, "loss": -0.0052, "grad_norm": 10.833090782165527, "learning_rate": 3.378787878787879e-06, "num_tokens": 4015681.0, "completions/mean_length": 56.5, "completions/min_length": 52.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.5, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9187343120574951, "rewards/meter/std": 0.15492095053195953, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9807487726211548, "rewards/repeat_soft/std": 0.01654265634715557, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.19255799055099487, "rewards/total_composite/mean": 0.8032553195953369, "rewards/total_composite/std": 0.0874934047460556, "reward": 0.8032553195953369, "reward_std": 0.087493397295475, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.142652228474617, "sampling/sampling_logp_difference/max": 2.502974510192871, "sampling/importance_sampling_ratio/min": 0.08184119313955307, "sampling/importance_sampling_ratio/mean": 0.9955335855484009, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7694462910294533, "clip_ratio/low_mean": 0.03276353236287832, "clip_ratio/low_min": 0.03276353236287832, "clip_ratio/high_mean": 0.09543735440820456, "clip_ratio/high_max": 0.09543735440820456, "clip_ratio/region_mean": 0.12820088677108288, "reward_total_mean": 0.8032553195953369, "reward_meter_mean": 0.9187343120574951, "reward_meter_std": 0.15492095053195953, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9807487726211548, "reward_repeat_soft_std": 0.01654265634715557, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.19255799055099487, "reward_total_composite_mean": 0.8032553195953369, "reward_total_composite_std": 0.0874934047460556} {"timestamp_utc": "2026-04-13T03:48:22Z", "mode": "train", "global_step": 2187, "epoch": 0.21968859869412355, "loss": 0.0614, "grad_norm": 9.160117149353027, "learning_rate": 3.375757575757576e-06, "num_tokens": 4018095.0, "completions/mean_length": 107.75, "completions/min_length": 94.0, "completions/max_length": 119.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.75, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.985000729560852, "rewards/meter/std": 0.0069869752041995525, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9830178022384644, "rewards/repeat_soft/std": 0.011045555584132671, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8175520896911621, "rewards/total_composite/std": 0.0032874627504497766, "reward": 0.8175520896911621, "reward_std": 0.0032874601893126965, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11918491125106812, "sampling/sampling_logp_difference/max": 1.426694393157959, "sampling/importance_sampling_ratio/min": 0.24010130763053894, "sampling/importance_sampling_ratio/mean": 0.9941781759262085, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6523789539933205, "clip_ratio/low_mean": 0.05031791562214494, "clip_ratio/low_min": 0.05031791562214494, "clip_ratio/high_mean": 0.07543424889445305, "clip_ratio/high_max": 0.07543424889445305, "clip_ratio/region_mean": 0.125752164516598, "reward_total_mean": 0.8175520896911621, "reward_meter_mean": 0.985000729560852, "reward_meter_std": 0.0069869752041995525, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9830178022384644, "reward_repeat_soft_std": 0.011045555584132671, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8175520896911621, "reward_total_composite_std": 0.0032874627504497766} {"timestamp_utc": "2026-04-13T03:48:29Z", "mode": "train", "global_step": 2188, "epoch": 0.21978905072827726, "loss": -0.0022, "grad_norm": 11.722826957702637, "learning_rate": 3.3727272727272732e-06, "num_tokens": 4019814.0, "completions/mean_length": 59.875, "completions/min_length": 53.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.875, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9680380821228027, "rewards/meter/std": 0.026508044451475143, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9821350574493408, "rewards/repeat_soft/std": 0.01284097507596016, "rewards/judge_quality/mean": 0.4962500333786011, "rewards/judge_quality/std": 0.17062386870384216, "rewards/total_composite/mean": 0.8327056169509888, "rewards/total_composite/std": 0.057838648557662964, "reward": 0.8327056169509888, "reward_std": 0.05783864110708237, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1120610237121582, "sampling/sampling_logp_difference/max": 1.8781356811523438, "sampling/importance_sampling_ratio/min": 0.15287485718727112, "sampling/importance_sampling_ratio/mean": 1.0208412408828735, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7326013371348381, "clip_ratio/low_mean": 0.07601392827928066, "clip_ratio/low_min": 0.07601392827928066, "clip_ratio/high_mean": 0.017537665087729692, "clip_ratio/high_max": 0.017537665087729692, "clip_ratio/region_mean": 0.09355159336701035, "reward_total_mean": 0.8327056169509888, "reward_meter_mean": 0.9680380821228027, "reward_meter_std": 0.026508044451475143, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9821350574493408, "reward_repeat_soft_std": 0.01284097507596016, "reward_judge_quality_mean": 0.4962500333786011, "reward_judge_quality_std": 0.17062386870384216, "reward_total_composite_mean": 0.8327056169509888, "reward_total_composite_std": 0.057838648557662964} {"timestamp_utc": "2026-04-13T03:48:36Z", "mode": "train", "global_step": 2189, "epoch": 0.21988950276243094, "loss": 0.0213, "grad_norm": 7.815253257751465, "learning_rate": 3.36969696969697e-06, "num_tokens": 4022171.0, "completions/mean_length": 115.625, "completions/min_length": 107.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.625, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.877539336681366, "rewards/meter/std": 0.1290019303560257, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9727838635444641, "rewards/repeat_soft/std": 0.02065858617424965, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7681710720062256, "rewards/total_composite/std": 0.05711708962917328, "reward": 0.7681710720062256, "reward_std": 0.057117074728012085, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11219421774148941, "sampling/sampling_logp_difference/max": 1.690810203552246, "sampling/importance_sampling_ratio/min": 0.18437008559703827, "sampling/importance_sampling_ratio/mean": 1.0165337324142456, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.620776079595089, "clip_ratio/low_mean": 0.05155478324741125, "clip_ratio/low_min": 0.05155478324741125, "clip_ratio/high_mean": 0.06194754224270582, "clip_ratio/high_max": 0.06194754224270582, "clip_ratio/region_mean": 0.11350232549011707, "reward_total_mean": 0.7681710720062256, "reward_meter_mean": 0.877539336681366, "reward_meter_std": 0.1290019303560257, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9727838635444641, "reward_repeat_soft_std": 0.02065858617424965, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7681710720062256, "reward_total_composite_std": 0.05711708962917328} {"timestamp_utc": "2026-04-13T03:48:42Z", "mode": "train", "global_step": 2190, "epoch": 0.21998995479658462, "loss": 0.0217, "grad_norm": 12.115137100219727, "learning_rate": 3.366666666666667e-06, "num_tokens": 4023666.0, "completions/mean_length": 27.875, "completions/min_length": 26.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.875, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.5758464336395264, "rewards/meter/std": 0.41496142745018005, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9503142833709717, "rewards/repeat_soft/std": 0.024309491738677025, "rewards/judge_quality/mean": 0.9200000166893005, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7801623344421387, "rewards/total_composite/std": 0.18573036789894104, "reward": 0.7801623344421387, "reward_std": 0.18573036789894104, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1073179617524147, "sampling/sampling_logp_difference/max": 1.4506378173828125, "sampling/importance_sampling_ratio/min": 0.23442071676254272, "sampling/importance_sampling_ratio/mean": 1.0166593790054321, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.817936897277832, "clip_ratio/low_mean": 0.0497311819344759, "clip_ratio/low_min": 0.0497311819344759, "clip_ratio/high_mean": 0.05783440778031945, "clip_ratio/high_max": 0.05783440778031945, "clip_ratio/region_mean": 0.10756558971479535, "reward_total_mean": 0.7801623344421387, "reward_meter_mean": 0.5758464336395264, "reward_meter_std": 0.41496142745018005, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9503142833709717, "reward_repeat_soft_std": 0.024309491738677025, "reward_judge_quality_mean": 0.9200000166893005, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7801623344421387, "reward_total_composite_std": 0.18573036789894104} {"timestamp_utc": "2026-04-13T03:48:49Z", "mode": "train", "global_step": 2191, "epoch": 0.22009040683073833, "loss": 0.0775, "grad_norm": 13.097497940063477, "learning_rate": 3.3636363636363637e-06, "num_tokens": 4025528.0, "completions/mean_length": 55.75, "completions/min_length": 48.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.75, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.8709853887557983, "rewards/meter/std": 0.3342602849006653, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.943076491355896, "rewards/repeat_soft/std": 0.07096266746520996, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.7603760361671448, "rewards/total_composite/std": 0.15125317871570587, "reward": 0.7603760361671448, "reward_std": 0.15125319361686707, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1165362074971199, "sampling/sampling_logp_difference/max": 2.3433265686035156, "sampling/importance_sampling_ratio/min": 0.09600772708654404, "sampling/importance_sampling_ratio/mean": 0.9996261596679688, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6069903820753098, "clip_ratio/low_mean": 0.004032257944345474, "clip_ratio/low_min": 0.004032257944345474, "clip_ratio/high_mean": 0.08484023809432983, "clip_ratio/high_max": 0.08484023809432983, "clip_ratio/region_mean": 0.08887249603867531, "reward_total_mean": 0.7603760361671448, "reward_meter_mean": 0.8709853887557983, "reward_meter_std": 0.3342602849006653, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.943076491355896, "reward_repeat_soft_std": 0.07096266746520996, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.7603760361671448, "reward_total_composite_std": 0.15125317871570587} {"timestamp_utc": "2026-04-13T03:49:00Z", "mode": "train", "global_step": 2192, "epoch": 0.220190858864892, "loss": -0.1082, "grad_norm": 1.8268153667449951, "learning_rate": 3.360606060606061e-06, "num_tokens": 4026961.0, "completions/mean_length": 93.125, "completions/min_length": 29.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 33.28571701049805, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9275498986244202, "rewards/meter/std": 0.09997802972793579, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.940275251865387, "rewards/repeat_soft/std": 0.04421278089284897, "rewards/judge_quality/mean": 0.36374998092651367, "rewards/judge_quality/std": 0.14302222430706024, "rewards/total_composite/mean": 0.6933602094650269, "rewards/total_composite/std": 0.28294575214385986, "reward": 0.6933602094650269, "reward_std": 0.28294575214385986, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12433776259422302, "sampling/sampling_logp_difference/max": 1.69037663936615, "sampling/importance_sampling_ratio/min": 0.18445004522800446, "sampling/importance_sampling_ratio/mean": 1.0210574865341187, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8307778835296631, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.15274546388536692, "clip_ratio/high_max": 0.15274546388536692, "clip_ratio/region_mean": 0.15274546388536692, "reward_total_mean": 0.6933602094650269, "reward_meter_mean": 0.9275498986244202, "reward_meter_std": 0.09997802972793579, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.940275251865387, "reward_repeat_soft_std": 0.04421278089284897, "reward_judge_quality_mean": 0.36374998092651367, "reward_judge_quality_std": 0.14302222430706024, "reward_total_composite_mean": 0.6933602094650269, "reward_total_composite_std": 0.28294575214385986} {"timestamp_utc": "2026-04-13T03:49:10Z", "mode": "train", "global_step": 2193, "epoch": 0.22029131089904572, "loss": 0.0496, "grad_norm": 5.528934001922607, "learning_rate": 3.357575757575758e-06, "num_tokens": 4029820.0, "completions/mean_length": 163.375, "completions/min_length": 148.0, "completions/max_length": 182.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 163.375, "completions/min_terminated_length": 148.0, "completions/max_terminated_length": 182.0, "rewards/meter/mean": 0.9754348993301392, "rewards/meter/std": 0.023764988407492638, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9463186264038086, "rewards/repeat_soft/std": 0.029847050085663795, "rewards/judge_quality/mean": 0.36000001430511475, "rewards/judge_quality/std": 0.1776030957698822, "rewards/total_composite/mean": 0.7878275513648987, "rewards/total_composite/std": 0.05576730892062187, "reward": 0.7878275513648987, "reward_std": 0.05576729774475098, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10940374433994293, "sampling/sampling_logp_difference/max": 2.823500871658325, "sampling/importance_sampling_ratio/min": 0.05939763784408569, "sampling/importance_sampling_ratio/mean": 1.0140957832336426, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7269374057650566, "clip_ratio/low_mean": 0.05372512200847268, "clip_ratio/low_min": 0.05372512200847268, "clip_ratio/high_mean": 0.0567833362147212, "clip_ratio/high_max": 0.0567833362147212, "clip_ratio/region_mean": 0.11050845822319388, "reward_total_mean": 0.7878275513648987, "reward_meter_mean": 0.9754348993301392, "reward_meter_std": 0.023764988407492638, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9463186264038086, "reward_repeat_soft_std": 0.029847050085663795, "reward_judge_quality_mean": 0.36000001430511475, "reward_judge_quality_std": 0.1776030957698822, "reward_total_composite_mean": 0.7878275513648987, "reward_total_composite_std": 0.05576730892062187} {"timestamp_utc": "2026-04-13T03:49:17Z", "mode": "train", "global_step": 2194, "epoch": 0.2203917629331994, "loss": 0.0208, "grad_norm": 7.4679107666015625, "learning_rate": 3.3545454545454547e-06, "num_tokens": 4031985.0, "completions/mean_length": 105.625, "completions/min_length": 95.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.625, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.8977329730987549, "rewards/meter/std": 0.16010113060474396, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9100542068481445, "rewards/repeat_soft/std": 0.0765482634305954, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.7544852495193481, "rewards/total_composite/std": 0.0818738043308258, "reward": 0.7544852495193481, "reward_std": 0.08187379688024521, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10904353111982346, "sampling/sampling_logp_difference/max": 2.5818331241607666, "sampling/importance_sampling_ratio/min": 0.07563523203134537, "sampling/importance_sampling_ratio/mean": 0.998033881187439, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5107659958302975, "clip_ratio/low_mean": 0.033780790865421295, "clip_ratio/low_min": 0.033780790865421295, "clip_ratio/high_mean": 0.06383551610633731, "clip_ratio/high_max": 0.06383551610633731, "clip_ratio/region_mean": 0.0976163069717586, "reward_total_mean": 0.7544852495193481, "reward_meter_mean": 0.8977329730987549, "reward_meter_std": 0.16010113060474396, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9100542068481445, "reward_repeat_soft_std": 0.0765482634305954, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.7544852495193481, "reward_total_composite_std": 0.0818738043308258} {"timestamp_utc": "2026-04-13T03:49:24Z", "mode": "train", "global_step": 2195, "epoch": 0.22049221496735308, "loss": 0.0628, "grad_norm": 10.173434257507324, "learning_rate": 3.351515151515152e-06, "num_tokens": 4034133.0, "completions/mean_length": 84.5, "completions/min_length": 77.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.5, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.8640680313110352, "rewards/meter/std": 0.2033507078886032, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9611980319023132, "rewards/repeat_soft/std": 0.016818564385175705, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7609504461288452, "rewards/total_composite/std": 0.09104262292385101, "reward": 0.7609504461288452, "reward_std": 0.0910426452755928, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11414583027362823, "sampling/sampling_logp_difference/max": 1.7024236917495728, "sampling/importance_sampling_ratio/min": 0.182241290807724, "sampling/importance_sampling_ratio/mean": 1.0236297845840454, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6998339667916298, "clip_ratio/low_mean": 0.024950592778623104, "clip_ratio/low_min": 0.024950592778623104, "clip_ratio/high_mean": 0.07992624584585428, "clip_ratio/high_max": 0.07992624584585428, "clip_ratio/region_mean": 0.10487683862447739, "reward_total_mean": 0.7609504461288452, "reward_meter_mean": 0.8640680313110352, "reward_meter_std": 0.2033507078886032, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9611980319023132, "reward_repeat_soft_std": 0.016818564385175705, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7609504461288452, "reward_total_composite_std": 0.09104262292385101} {"timestamp_utc": "2026-04-13T03:49:31Z", "mode": "train", "global_step": 2196, "epoch": 0.2205926670015068, "loss": 0.0587, "grad_norm": 9.607287406921387, "learning_rate": 3.3484848484848487e-06, "num_tokens": 4035906.0, "completions/mean_length": 56.625, "completions/min_length": 49.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.625, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.8607462644577026, "rewards/meter/std": 0.16724137961864471, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9882793426513672, "rewards/repeat_soft/std": 0.004271867219358683, "rewards/judge_quality/mean": 0.518750011920929, "rewards/judge_quality/std": 0.1751275211572647, "rewards/total_composite/mean": 0.7917886972427368, "rewards/total_composite/std": 0.10783903300762177, "reward": 0.7917886972427368, "reward_std": 0.10783904045820236, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09751703590154648, "sampling/sampling_logp_difference/max": 0.8970643281936646, "sampling/importance_sampling_ratio/min": 0.4077649712562561, "sampling/importance_sampling_ratio/mean": 1.000393033027649, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6136078014969826, "clip_ratio/low_mean": 0.03634570399299264, "clip_ratio/low_min": 0.03634570399299264, "clip_ratio/high_mean": 0.06841218797490001, "clip_ratio/high_max": 0.06841218797490001, "clip_ratio/region_mean": 0.10475789196789265, "reward_total_mean": 0.7917886972427368, "reward_meter_mean": 0.8607462644577026, "reward_meter_std": 0.16724137961864471, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9882793426513672, "reward_repeat_soft_std": 0.004271867219358683, "reward_judge_quality_mean": 0.518750011920929, "reward_judge_quality_std": 0.1751275211572647, "reward_total_composite_mean": 0.7917886972427368, "reward_total_composite_std": 0.10783903300762177} {"timestamp_utc": "2026-04-13T03:49:38Z", "mode": "train", "global_step": 2197, "epoch": 0.22069311903566047, "loss": 0.0179, "grad_norm": 10.662117958068848, "learning_rate": 3.3454545454545456e-06, "num_tokens": 4037536.0, "completions/mean_length": 56.75, "completions/min_length": 54.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.75, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9106227159500122, "rewards/meter/std": 0.18485231697559357, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.996672511100769, "rewards/repeat_soft/std": 0.006435038056224585, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7910724878311157, "rewards/total_composite/std": 0.08173540979623795, "reward": 0.7910724878311157, "reward_std": 0.08173539489507675, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08876968175172806, "sampling/sampling_logp_difference/max": 1.0318522453308105, "sampling/importance_sampling_ratio/min": 0.3563463091850281, "sampling/importance_sampling_ratio/mean": 1.0137425661087036, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.515534907579422, "clip_ratio/low_mean": 0.008620689623057842, "clip_ratio/low_min": 0.008620689623057842, "clip_ratio/high_mean": 0.0777452765032649, "clip_ratio/high_max": 0.0777452765032649, "clip_ratio/region_mean": 0.08636596612632275, "reward_total_mean": 0.7910724878311157, "reward_meter_mean": 0.9106227159500122, "reward_meter_std": 0.18485231697559357, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.996672511100769, "reward_repeat_soft_std": 0.006435038056224585, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7910724878311157, "reward_total_composite_std": 0.08173540979623795} {"timestamp_utc": "2026-04-13T03:49:46Z", "mode": "train", "global_step": 2198, "epoch": 0.22079357106981418, "loss": 0.0019, "grad_norm": 9.656511306762695, "learning_rate": 3.3424242424242424e-06, "num_tokens": 4039859.0, "completions/mean_length": 103.375, "completions/min_length": 94.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.375, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.8129128813743591, "rewards/meter/std": 0.21503625810146332, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9046807289123535, "rewards/repeat_soft/std": 0.09675776958465576, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.6767789125442505, "rewards/total_composite/std": 0.09642530977725983, "reward": 0.6767789125442505, "reward_std": 0.09642530977725983, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13095778226852417, "sampling/sampling_logp_difference/max": 3.676079034805298, "sampling/importance_sampling_ratio/min": 0.025322068482637405, "sampling/importance_sampling_ratio/mean": 1.0068068504333496, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5279671885073185, "clip_ratio/low_mean": 0.01967031811363995, "clip_ratio/low_min": 0.01967031811363995, "clip_ratio/high_mean": 0.09010688308626413, "clip_ratio/high_max": 0.09010688308626413, "clip_ratio/region_mean": 0.10977720119990408, "reward_total_mean": 0.6767789125442505, "reward_meter_mean": 0.8129128813743591, "reward_meter_std": 0.21503625810146332, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9046807289123535, "reward_repeat_soft_std": 0.09675776958465576, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.6767789125442505, "reward_total_composite_std": 0.09642530977725983} {"timestamp_utc": "2026-04-13T03:49:57Z", "mode": "train", "global_step": 2199, "epoch": 0.22089402310396786, "loss": -0.1124, "grad_norm": 3.0926787853240967, "learning_rate": 3.3393939393939397e-06, "num_tokens": 4041503.0, "completions/mean_length": 113.5, "completions/min_length": 53.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 56.57143020629883, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.7500079870223999, "rewards/meter/std": 0.37415412068367004, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9691245555877686, "rewards/repeat_soft/std": 0.023802226409316063, "rewards/judge_quality/mean": 0.3712500035762787, "rewards/judge_quality/std": 0.217547208070755, "rewards/total_composite/mean": 0.6088094711303711, "rewards/total_composite/std": 0.2987876832485199, "reward": 0.6088094711303711, "reward_std": 0.2987876832485199, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.142760768532753, "sampling/sampling_logp_difference/max": 2.4570515155792236, "sampling/importance_sampling_ratio/min": 0.08568722754716873, "sampling/importance_sampling_ratio/mean": 1.0060014724731445, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6669965609908104, "clip_ratio/low_mean": 0.03190789557993412, "clip_ratio/low_min": 0.03190789557993412, "clip_ratio/high_mean": 0.08479883149266243, "clip_ratio/high_max": 0.08479883149266243, "clip_ratio/region_mean": 0.11670672707259655, "reward_total_mean": 0.6088094711303711, "reward_meter_mean": 0.7500079870223999, "reward_meter_std": 0.37415412068367004, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9691245555877686, "reward_repeat_soft_std": 0.023802226409316063, "reward_judge_quality_mean": 0.3712500035762787, "reward_judge_quality_std": 0.217547208070755, "reward_total_composite_mean": 0.6088094711303711, "reward_total_composite_std": 0.2987876832485199} {"timestamp_utc": "2026-04-13T03:50:04Z", "mode": "train", "global_step": 2200, "epoch": 0.22099447513812154, "loss": 0.0401, "grad_norm": 11.532222747802734, "learning_rate": 3.3363636363636365e-06, "num_tokens": 4043273.0, "completions/mean_length": 51.25, "completions/min_length": 47.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.25, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9119675159454346, "rewards/meter/std": 0.11493070423603058, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9934697151184082, "rewards/repeat_soft/std": 0.006171245127916336, "rewards/judge_quality/mean": 0.9200000166893005, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.935732364654541, "rewards/total_composite/std": 0.051677241921424866, "reward": 0.935732364654541, "reward_std": 0.051677241921424866, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13416609168052673, "sampling/sampling_logp_difference/max": 3.4758434295654297, "sampling/importance_sampling_ratio/min": 0.030935728922486305, "sampling/importance_sampling_ratio/mean": 0.9872518181800842, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7837947830557823, "clip_ratio/low_mean": 0.027867589611560106, "clip_ratio/low_min": 0.027867589611560106, "clip_ratio/high_mean": 0.07262203563004732, "clip_ratio/high_max": 0.07262203563004732, "clip_ratio/region_mean": 0.10048962524160743, "reward_total_mean": 0.935732364654541, "reward_meter_mean": 0.9119675159454346, "reward_meter_std": 0.11493070423603058, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9934697151184082, "reward_repeat_soft_std": 0.006171245127916336, "reward_judge_quality_mean": 0.9200000166893005, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.935732364654541, "reward_total_composite_std": 0.051677241921424866} {"timestamp_utc": "2026-04-13T03:51:06Z", "mode": "eval", "global_step": 2200, "epoch": 0.22099447513812154, "eval_loss": NaN, "eval_runtime": 62.3957, "eval_samples_per_second": 1.282, "eval_steps_per_second": 0.16, "eval_num_tokens": 4043273.0, "eval_completions/mean_length": 112.5875, "eval_completions/min_length": 40.4, "eval_completions/max_length": 299.7, "eval_completions/clipped_ratio": 0.05, "eval_completions/mean_terminated_length": 91.39107284545898, "eval_completions/min_terminated_length": 40.4, "eval_completions/max_terminated_length": 149.1, "eval_rewards/meter/mean": 0.8306773960590362, "eval_rewards/meter/std": 0.3041162356734276, "eval_rewards/count_adherence/mean": 0.9445833444595337, "eval_rewards/count_adherence/std": 0.12017041519284248, "eval_rewards/hard_gate/mean": 0.9625, "eval_rewards/hard_gate/std": 0.10606601536273956, "eval_rewards/repeat_soft/mean": 0.9619694113731384, "eval_rewards/repeat_soft/std": 0.0365039786323905, "eval_rewards/judge_quality/mean": 0.45175000429153445, "eval_rewards/judge_quality/std": 0.1769706428050995, "eval_rewards/total_composite/mean": 0.7321358501911164, "eval_rewards/total_composite/std": 0.1891805797815323, "eval_reward": 0.7321358501911164, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.06320357620716095, "eval_sampling/sampling_logp_difference/max": 0.9940471172332763, "eval_sampling/importance_sampling_ratio/min": 0.3766573786735535, "eval_sampling/importance_sampling_ratio/mean": 1.0134490370750426, "eval_sampling/importance_sampling_ratio/max": 1.5173115372657775, "eval_entropy": 0.6875483095645905, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7321358501911164, "eval_reward_meter_mean": 0.8306773960590362, "eval_reward_meter_std": 0.3041162356734276, "eval_reward_count_adherence_mean": 0.9445833444595337, "eval_reward_count_adherence_std": 0.12017041519284248, "eval_reward_hard_gate_mean": 0.9625, "eval_reward_hard_gate_std": 0.10606601536273956, "eval_reward_repeat_soft_mean": 0.9619694113731384, "eval_reward_repeat_soft_std": 0.0365039786323905, "eval_reward_judge_quality_mean": 0.45175000429153445, "eval_reward_judge_quality_std": 0.1769706428050995, "eval_reward_total_composite_mean": 0.7321358501911164, "eval_reward_total_composite_std": 0.1891805797815323} {"timestamp_utc": "2026-04-13T03:51:21Z", "mode": "train", "global_step": 2201, "epoch": 0.22109492717227525, "loss": -0.1346, "grad_norm": 3.331361770629883, "learning_rate": 3.3333333333333333e-06, "num_tokens": 4044925.0, "completions/mean_length": 111.5, "completions/min_length": 45.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 54.28571701049805, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.8260104060173035, "rewards/meter/std": 0.29196488857269287, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9792269468307495, "rewards/repeat_soft/std": 0.012598730623722076, "rewards/judge_quality/mean": 0.7487499713897705, "rewards/judge_quality/std": 0.332154780626297, "rewards/total_composite/mean": 0.7561932802200317, "rewards/total_composite/std": 0.35054251551628113, "reward": 0.7561932802200317, "reward_std": 0.35054251551628113, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12169025093317032, "sampling/sampling_logp_difference/max": 3.3953044414520264, "sampling/importance_sampling_ratio/min": 0.033530343323946, "sampling/importance_sampling_ratio/mean": 1.0082714557647705, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4888042211532593, "clip_ratio/low_mean": 0.028975095599889755, "clip_ratio/low_min": 0.028975095599889755, "clip_ratio/high_mean": 0.07702233456075191, "clip_ratio/high_max": 0.07702233456075191, "clip_ratio/region_mean": 0.10599743016064167, "reward_total_mean": 0.7561932802200317, "reward_meter_mean": 0.8260104060173035, "reward_meter_std": 0.29196488857269287, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9792269468307495, "reward_repeat_soft_std": 0.012598730623722076, "reward_judge_quality_mean": 0.7487499713897705, "reward_judge_quality_std": 0.332154780626297, "reward_total_composite_mean": 0.7561932802200317, "reward_total_composite_std": 0.35054251551628113} {"timestamp_utc": "2026-04-13T03:51:32Z", "mode": "train", "global_step": 2202, "epoch": 0.22119537920642893, "loss": -0.1894, "grad_norm": 2.193657875061035, "learning_rate": 3.3303030303030306e-06, "num_tokens": 4046923.0, "completions/mean_length": 143.75, "completions/min_length": 81.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 91.14286041259766, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.9720015525817871, "rewards/meter/std": 0.028503766283392906, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9518323540687561, "rewards/repeat_soft/std": 0.02955063246190548, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.7101519703865051, "rewards/total_composite/std": 0.28701654076576233, "reward": 0.7101519703865051, "reward_std": 0.28701654076576233, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12211467325687408, "sampling/sampling_logp_difference/max": 2.9326424598693848, "sampling/importance_sampling_ratio/min": 0.05325612425804138, "sampling/importance_sampling_ratio/mean": 1.0093281269073486, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.61939537525177, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09577116277068853, "clip_ratio/high_max": 0.09577116277068853, "clip_ratio/region_mean": 0.09577116277068853, "reward_total_mean": 0.7101519703865051, "reward_meter_mean": 0.9720015525817871, "reward_meter_std": 0.028503766283392906, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9518323540687561, "reward_repeat_soft_std": 0.02955063246190548, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.7101519703865051, "reward_total_composite_std": 0.28701654076576233} {"timestamp_utc": "2026-04-13T03:51:39Z", "mode": "train", "global_step": 2203, "epoch": 0.22129583124058264, "loss": -0.045, "grad_norm": 8.48707389831543, "learning_rate": 3.3272727272727274e-06, "num_tokens": 4048677.0, "completions/mean_length": 64.25, "completions/min_length": 53.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.25, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9450699090957642, "rewards/meter/std": 0.07073565572500229, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9898174405097961, "rewards/repeat_soft/std": 0.010681229643523693, "rewards/judge_quality/mean": 0.42624998092651367, "rewards/judge_quality/std": 0.14647647738456726, "rewards/total_composite/mean": 0.69875568151474, "rewards/total_composite/std": 0.2842884063720703, "reward": 0.69875568151474, "reward_std": 0.2842884063720703, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1307879090309143, "sampling/sampling_logp_difference/max": 1.65114164352417, "sampling/importance_sampling_ratio/min": 0.19183078408241272, "sampling/importance_sampling_ratio/mean": 1.0138341188430786, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8278946131467819, "clip_ratio/low_mean": 0.021226415410637856, "clip_ratio/low_min": 0.021226415410637856, "clip_ratio/high_mean": 0.09548099199309945, "clip_ratio/high_max": 0.09548099199309945, "clip_ratio/region_mean": 0.1167074074037373, "reward_total_mean": 0.69875568151474, "reward_meter_mean": 0.9450699090957642, "reward_meter_std": 0.07073565572500229, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9898174405097961, "reward_repeat_soft_std": 0.010681229643523693, "reward_judge_quality_mean": 0.42624998092651367, "reward_judge_quality_std": 0.14647647738456726, "reward_total_composite_mean": 0.69875568151474, "reward_total_composite_std": 0.2842884063720703} {"timestamp_utc": "2026-04-13T03:51:46Z", "mode": "train", "global_step": 2204, "epoch": 0.22139628327473632, "loss": 0.0228, "grad_norm": 8.338180541992188, "learning_rate": 3.3242424242424242e-06, "num_tokens": 4051005.0, "completions/mean_length": 100.0, "completions/min_length": 91.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.0, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.9840829968452454, "rewards/meter/std": 0.0062208641320466995, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9229989051818848, "rewards/repeat_soft/std": 0.09551280736923218, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.8047622442245483, "rewards/total_composite/std": 0.017010590061545372, "reward": 0.8047622442245483, "reward_std": 0.01701059564948082, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.136907160282135, "sampling/sampling_logp_difference/max": 2.7469606399536133, "sampling/importance_sampling_ratio/min": 0.06412246078252792, "sampling/importance_sampling_ratio/mean": 0.9921556115150452, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7195044830441475, "clip_ratio/low_mean": 0.03220982104539871, "clip_ratio/low_min": 0.03220982104539871, "clip_ratio/high_mean": 0.08489525970071554, "clip_ratio/high_max": 0.08489525970071554, "clip_ratio/region_mean": 0.11710508074611425, "reward_total_mean": 0.8047622442245483, "reward_meter_mean": 0.9840829968452454, "reward_meter_std": 0.0062208641320466995, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9229989051818848, "reward_repeat_soft_std": 0.09551280736923218, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.8047622442245483, "reward_total_composite_std": 0.017010590061545372} {"timestamp_utc": "2026-04-13T03:51:53Z", "mode": "train", "global_step": 2205, "epoch": 0.22149673530889, "loss": -0.0519, "grad_norm": 9.265081405639648, "learning_rate": 3.321212121212121e-06, "num_tokens": 4052914.0, "completions/mean_length": 58.625, "completions/min_length": 49.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.625, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9040992259979248, "rewards/meter/std": 0.10776462405920029, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9794857501983643, "rewards/repeat_soft/std": 0.018766749650239944, "rewards/judge_quality/mean": 0.7150000333786011, "rewards/judge_quality/std": 0.2887411117553711, "rewards/total_composite/mean": 0.869293212890625, "rewards/total_composite/std": 0.118052177131176, "reward": 0.869293212890625, "reward_std": 0.118052177131176, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12251515686511993, "sampling/sampling_logp_difference/max": 2.0451388359069824, "sampling/importance_sampling_ratio/min": 0.12936222553253174, "sampling/importance_sampling_ratio/mean": 1.015015959739685, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8463640213012695, "clip_ratio/low_mean": 0.04039201978594065, "clip_ratio/low_min": 0.04039201978594065, "clip_ratio/high_mean": 0.08818680513650179, "clip_ratio/high_max": 0.08818680513650179, "clip_ratio/region_mean": 0.12857882492244244, "reward_total_mean": 0.869293212890625, "reward_meter_mean": 0.9040992259979248, "reward_meter_std": 0.10776462405920029, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9794857501983643, "reward_repeat_soft_std": 0.018766749650239944, "reward_judge_quality_mean": 0.7150000333786011, "reward_judge_quality_std": 0.2887411117553711, "reward_total_composite_mean": 0.869293212890625, "reward_total_composite_std": 0.118052177131176} {"timestamp_utc": "2026-04-13T03:52:01Z", "mode": "train", "global_step": 2206, "epoch": 0.2215971873430437, "loss": 0.0338, "grad_norm": 8.493517875671387, "learning_rate": 3.3181818181818188e-06, "num_tokens": 4055539.0, "completions/mean_length": 137.125, "completions/min_length": 123.0, "completions/max_length": 149.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 137.125, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 149.0, "rewards/meter/mean": 0.8994851112365723, "rewards/meter/std": 0.24052610993385315, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9598791599273682, "rewards/repeat_soft/std": 0.010929180309176445, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.7816312313079834, "rewards/total_composite/std": 0.13558581471443176, "reward": 0.7816312313079834, "reward_std": 0.13558582961559296, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13067780435085297, "sampling/sampling_logp_difference/max": 2.305039405822754, "sampling/importance_sampling_ratio/min": 0.09975486248731613, "sampling/importance_sampling_ratio/mean": 1.0042474269866943, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8429490998387337, "clip_ratio/low_mean": 0.01708633080124855, "clip_ratio/low_min": 0.01708633080124855, "clip_ratio/high_mean": 0.11858692299574614, "clip_ratio/high_max": 0.11858692299574614, "clip_ratio/region_mean": 0.1356732537969947, "reward_total_mean": 0.7816312313079834, "reward_meter_mean": 0.8994851112365723, "reward_meter_std": 0.24052610993385315, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9598791599273682, "reward_repeat_soft_std": 0.010929180309176445, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.7816312313079834, "reward_total_composite_std": 0.13558581471443176} {"timestamp_utc": "2026-04-13T03:52:07Z", "mode": "train", "global_step": 2207, "epoch": 0.2216976393771974, "loss": 0.1818, "grad_norm": 17.19708251953125, "learning_rate": 3.3151515151515156e-06, "num_tokens": 4057142.0, "completions/mean_length": 49.375, "completions/min_length": 38.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.375, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.7736392021179199, "rewards/meter/std": 0.26332297921180725, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9918738603591919, "rewards/repeat_soft/std": 0.006677909288555384, "rewards/judge_quality/mean": 0.6225000023841858, "rewards/judge_quality/std": 0.24656210839748383, "rewards/total_composite/mean": 0.7653250694274902, "rewards/total_composite/std": 0.14413821697235107, "reward": 0.7653250694274902, "reward_std": 0.14413821697235107, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12349814921617508, "sampling/sampling_logp_difference/max": 1.1094884872436523, "sampling/importance_sampling_ratio/min": 0.3370956778526306, "sampling/importance_sampling_ratio/mean": 1.0148710012435913, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6739152371883392, "clip_ratio/low_mean": 0.053750863298773766, "clip_ratio/low_min": 0.053750863298773766, "clip_ratio/high_mean": 0.07436970435082912, "clip_ratio/high_max": 0.07436970435082912, "clip_ratio/region_mean": 0.1281205676496029, "reward_total_mean": 0.7653250694274902, "reward_meter_mean": 0.7736392021179199, "reward_meter_std": 0.26332297921180725, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9918738603591919, "reward_repeat_soft_std": 0.006677909288555384, "reward_judge_quality_mean": 0.6225000023841858, "reward_judge_quality_std": 0.24656210839748383, "reward_total_composite_mean": 0.7653250694274902, "reward_total_composite_std": 0.14413821697235107} {"timestamp_utc": "2026-04-13T03:52:15Z", "mode": "train", "global_step": 2208, "epoch": 0.22179809141135107, "loss": 0.0344, "grad_norm": 8.720504760742188, "learning_rate": 3.3121212121212124e-06, "num_tokens": 4059415.0, "completions/mean_length": 104.125, "completions/min_length": 92.0, "completions/max_length": 110.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.125, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 110.0, "rewards/meter/mean": 0.9337171912193298, "rewards/meter/std": 0.10564862191677094, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.981008768081665, "rewards/repeat_soft/std": 0.01431102491915226, "rewards/judge_quality/mean": 0.6449999809265137, "rewards/judge_quality/std": 0.2121320515871048, "rewards/total_composite/mean": 0.8617736101150513, "rewards/total_composite/std": 0.06928357481956482, "reward": 0.8617736101150513, "reward_std": 0.06928356736898422, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12986770272254944, "sampling/sampling_logp_difference/max": 2.8148975372314453, "sampling/importance_sampling_ratio/min": 0.059910859912633896, "sampling/importance_sampling_ratio/mean": 1.0019067525863647, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7242220863699913, "clip_ratio/low_mean": 0.04482605308294296, "clip_ratio/low_min": 0.04482605308294296, "clip_ratio/high_mean": 0.05690174736082554, "clip_ratio/high_max": 0.05690174736082554, "clip_ratio/region_mean": 0.1017278004437685, "reward_total_mean": 0.8617736101150513, "reward_meter_mean": 0.9337171912193298, "reward_meter_std": 0.10564862191677094, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.981008768081665, "reward_repeat_soft_std": 0.01431102491915226, "reward_judge_quality_mean": 0.6449999809265137, "reward_judge_quality_std": 0.2121320515871048, "reward_total_composite_mean": 0.8617736101150513, "reward_total_composite_std": 0.06928357481956482} {"timestamp_utc": "2026-04-13T03:52:21Z", "mode": "train", "global_step": 2209, "epoch": 0.22189854344550478, "loss": 0.0367, "grad_norm": 12.024104118347168, "learning_rate": 3.3090909090909097e-06, "num_tokens": 4061106.0, "completions/mean_length": 44.375, "completions/min_length": 38.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.375, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.9134516716003418, "rewards/meter/std": 0.0807996466755867, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9873526692390442, "rewards/repeat_soft/std": 0.0166508499532938, "rewards/judge_quality/mean": 0.5349999666213989, "rewards/judge_quality/std": 0.18431341648101807, "rewards/total_composite/mean": 0.8202884793281555, "rewards/total_composite/std": 0.07605648785829544, "reward": 0.8202884793281555, "reward_std": 0.07605647295713425, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11195176839828491, "sampling/sampling_logp_difference/max": 1.137956142425537, "sampling/importance_sampling_ratio/min": 0.3204733431339264, "sampling/importance_sampling_ratio/mean": 0.9936956167221069, "sampling/importance_sampling_ratio/max": 1.9016095399856567, "entropy": 0.6828983053565025, "clip_ratio/low_mean": 0.08212634827941656, "clip_ratio/low_min": 0.08212634827941656, "clip_ratio/high_mean": 0.03394736908376217, "clip_ratio/high_max": 0.03394736908376217, "clip_ratio/region_mean": 0.11607371736317873, "reward_total_mean": 0.8202884793281555, "reward_meter_mean": 0.9134516716003418, "reward_meter_std": 0.0807996466755867, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9873526692390442, "reward_repeat_soft_std": 0.0166508499532938, "reward_judge_quality_mean": 0.5349999666213989, "reward_judge_quality_std": 0.18431341648101807, "reward_total_composite_mean": 0.8202884793281555, "reward_total_composite_std": 0.07605648785829544} {"timestamp_utc": "2026-04-13T03:52:33Z", "mode": "train", "global_step": 2210, "epoch": 0.22199899547965846, "loss": -0.1234, "grad_norm": 2.719426155090332, "learning_rate": 3.3060606060606065e-06, "num_tokens": 4062676.0, "completions/mean_length": 106.25, "completions/min_length": 45.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 48.28571701049805, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.785103440284729, "rewards/meter/std": 0.3610110580921173, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9062859416007996, "rewards/repeat_soft/std": 0.06781671941280365, "rewards/judge_quality/mean": 0.36374998092651367, "rewards/judge_quality/std": 0.14302222430706024, "rewards/total_composite/mean": 0.6471767425537109, "rewards/total_composite/std": 0.296779066324234, "reward": 0.6471767425537109, "reward_std": 0.296779066324234, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10379276424646378, "sampling/sampling_logp_difference/max": 1.6643764972686768, "sampling/importance_sampling_ratio/min": 0.18930864334106445, "sampling/importance_sampling_ratio/mean": 1.0039098262786865, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44165628775954247, "clip_ratio/low_mean": 0.013888888992369175, "clip_ratio/low_min": 0.013888888992369175, "clip_ratio/high_mean": 0.07095170579850674, "clip_ratio/high_max": 0.07095170579850674, "clip_ratio/region_mean": 0.08484059479087591, "reward_total_mean": 0.6471767425537109, "reward_meter_mean": 0.785103440284729, "reward_meter_std": 0.3610110580921173, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9062859416007996, "reward_repeat_soft_std": 0.06781671941280365, "reward_judge_quality_mean": 0.36374998092651367, "reward_judge_quality_std": 0.14302222430706024, "reward_total_composite_mean": 0.6471767425537109, "reward_total_composite_std": 0.296779066324234} {"timestamp_utc": "2026-04-13T03:52:39Z", "mode": "train", "global_step": 2211, "epoch": 0.22209944751381216, "loss": 0.111, "grad_norm": 9.250675201416016, "learning_rate": 3.3030303030303033e-06, "num_tokens": 4064384.0, "completions/mean_length": 52.5, "completions/min_length": 47.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.5, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9785383939743042, "rewards/meter/std": 0.013131825253367424, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9578072428703308, "rewards/repeat_soft/std": 0.039800334721803665, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8058729767799377, "rewards/total_composite/std": 0.01693149283528328, "reward": 0.8058729767799377, "reward_std": 0.016931507736444473, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08453433960676193, "sampling/sampling_logp_difference/max": 1.01739501953125, "sampling/importance_sampling_ratio/min": 0.36153551936149597, "sampling/importance_sampling_ratio/mean": 1.009893774986267, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.41661009192466736, "clip_ratio/low_mean": 0.024153543170541525, "clip_ratio/low_min": 0.024153543170541525, "clip_ratio/high_mean": 0.0406859174836427, "clip_ratio/high_max": 0.0406859174836427, "clip_ratio/region_mean": 0.06483946065418422, "reward_total_mean": 0.8058729767799377, "reward_meter_mean": 0.9785383939743042, "reward_meter_std": 0.013131825253367424, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9578072428703308, "reward_repeat_soft_std": 0.039800334721803665, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8058729767799377, "reward_total_composite_std": 0.01693149283528328} {"timestamp_utc": "2026-04-13T03:52:46Z", "mode": "train", "global_step": 2212, "epoch": 0.22219989954796585, "loss": 0.0218, "grad_norm": 8.91968059539795, "learning_rate": 3.3000000000000006e-06, "num_tokens": 4066371.0, "completions/mean_length": 79.375, "completions/min_length": 69.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.375, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.8950619101524353, "rewards/meter/std": 0.19598466157913208, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9859611988067627, "rewards/repeat_soft/std": 0.013018976897001266, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.19334924221038818, "rewards/total_composite/mean": 0.7919989824295044, "rewards/total_composite/std": 0.1140722930431366, "reward": 0.7919989824295044, "reward_std": 0.11407230794429779, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12338856607675552, "sampling/sampling_logp_difference/max": 2.664079427719116, "sampling/importance_sampling_ratio/min": 0.06966345012187958, "sampling/importance_sampling_ratio/mean": 1.0172606706619263, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7770102471113205, "clip_ratio/low_mean": 0.04518844746053219, "clip_ratio/low_min": 0.04518844746053219, "clip_ratio/high_mean": 0.08168669044971466, "clip_ratio/high_max": 0.08168669044971466, "clip_ratio/region_mean": 0.12687513791024685, "reward_total_mean": 0.7919989824295044, "reward_meter_mean": 0.8950619101524353, "reward_meter_std": 0.19598466157913208, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9859611988067627, "reward_repeat_soft_std": 0.013018976897001266, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.19334924221038818, "reward_total_composite_mean": 0.7919989824295044, "reward_total_composite_std": 0.1140722930431366} {"timestamp_utc": "2026-04-13T03:52:52Z", "mode": "train", "global_step": 2213, "epoch": 0.22230035158211953, "loss": 0.0592, "grad_norm": 11.513107299804688, "learning_rate": 3.2969696969696974e-06, "num_tokens": 4068093.0, "completions/mean_length": 58.25, "completions/min_length": 52.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.25, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9821536540985107, "rewards/meter/std": 0.013355078175663948, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9933594465255737, "rewards/repeat_soft/std": 0.010055365040898323, "rewards/judge_quality/mean": 0.39249998331069946, "rewards/judge_quality/std": 0.08892211318016052, "rewards/total_composite/mean": 0.8090550899505615, "rewards/total_composite/std": 0.030630676075816154, "reward": 0.8090550899505615, "reward_std": 0.030630672350525856, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11081297695636749, "sampling/sampling_logp_difference/max": 1.1517348289489746, "sampling/importance_sampling_ratio/min": 0.316087931394577, "sampling/importance_sampling_ratio/mean": 1.01006281375885, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7399673834443092, "clip_ratio/low_mean": 0.034050178714096546, "clip_ratio/low_min": 0.034050178714096546, "clip_ratio/high_mean": 0.08800942404195666, "clip_ratio/high_max": 0.08800942404195666, "clip_ratio/region_mean": 0.12205960275605321, "reward_total_mean": 0.8090550899505615, "reward_meter_mean": 0.9821536540985107, "reward_meter_std": 0.013355078175663948, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9933594465255737, "reward_repeat_soft_std": 0.010055365040898323, "reward_judge_quality_mean": 0.39249998331069946, "reward_judge_quality_std": 0.08892211318016052, "reward_total_composite_mean": 0.8090550899505615, "reward_total_composite_std": 0.030630676075816154} {"timestamp_utc": "2026-04-13T03:53:04Z", "mode": "train", "global_step": 2214, "epoch": 0.22240080361627323, "loss": -0.2636, "grad_norm": 4.0228776931762695, "learning_rate": 3.2939393939393943e-06, "num_tokens": 4070014.0, "completions/mean_length": 148.125, "completions/min_length": 10.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 96.14286041259766, "completions/min_terminated_length": 10.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.35800430178642273, "rewards/meter/std": 0.29404303431510925, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.41052016615867615, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9705147743225098, "rewards/repeat_soft/std": 0.021026551723480225, "rewards/judge_quality/mean": 0.3187499940395355, "rewards/judge_quality/std": 0.14961259067058563, "rewards/total_composite/mean": 0.4341520369052887, "rewards/total_composite/std": 0.28626611828804016, "reward": 0.4341520369052887, "reward_std": 0.28626611828804016, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12240725010633469, "sampling/sampling_logp_difference/max": 2.7280101776123047, "sampling/importance_sampling_ratio/min": 0.06534919142723083, "sampling/importance_sampling_ratio/mean": 0.9993200898170471, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7657416462898254, "clip_ratio/low_mean": 0.03019911516457796, "clip_ratio/low_min": 0.03019911516457796, "clip_ratio/high_mean": 0.07644016481935978, "clip_ratio/high_max": 0.07644016481935978, "clip_ratio/region_mean": 0.10663927998393774, "reward_total_mean": 0.4341520369052887, "reward_meter_mean": 0.35800430178642273, "reward_meter_std": 0.29404303431510925, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.41052016615867615, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9705147743225098, "reward_repeat_soft_std": 0.021026551723480225, "reward_judge_quality_mean": 0.3187499940395355, "reward_judge_quality_std": 0.14961259067058563, "reward_total_composite_mean": 0.4341520369052887, "reward_total_composite_std": 0.28626611828804016} {"timestamp_utc": "2026-04-13T03:53:10Z", "mode": "train", "global_step": 2215, "epoch": 0.22250125565042692, "loss": -0.0107, "grad_norm": 14.015032768249512, "learning_rate": 3.290909090909091e-06, "num_tokens": 4071680.0, "completions/mean_length": 49.25, "completions/min_length": 41.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.25, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9764039516448975, "rewards/meter/std": 0.037458647042512894, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9671276807785034, "rewards/repeat_soft/std": 0.027699442580342293, "rewards/judge_quality/mean": 0.6512500047683716, "rewards/judge_quality/std": 0.18542905151844025, "rewards/total_composite/mean": 0.8814696073532104, "rewards/total_composite/std": 0.06586789339780807, "reward": 0.8814696073532104, "reward_std": 0.06586787849664688, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1424378603696823, "sampling/sampling_logp_difference/max": 1.5785412788391113, "sampling/importance_sampling_ratio/min": 0.20627577602863312, "sampling/importance_sampling_ratio/mean": 0.9920380115509033, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8113629668951035, "clip_ratio/low_mean": 0.06637217849493027, "clip_ratio/low_min": 0.06637217849493027, "clip_ratio/high_mean": 0.08815397135913372, "clip_ratio/high_max": 0.08815397135913372, "clip_ratio/region_mean": 0.154526149854064, "reward_total_mean": 0.8814696073532104, "reward_meter_mean": 0.9764039516448975, "reward_meter_std": 0.037458647042512894, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9671276807785034, "reward_repeat_soft_std": 0.027699442580342293, "reward_judge_quality_mean": 0.6512500047683716, "reward_judge_quality_std": 0.18542905151844025, "reward_total_composite_mean": 0.8814696073532104, "reward_total_composite_std": 0.06586789339780807} {"timestamp_utc": "2026-04-13T03:53:16Z", "mode": "train", "global_step": 2216, "epoch": 0.22260170768458062, "loss": 0.0684, "grad_norm": 11.883666038513184, "learning_rate": 3.2878787878787883e-06, "num_tokens": 4073299.0, "completions/mean_length": 48.375, "completions/min_length": 45.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.375, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9861613512039185, "rewards/meter/std": 0.005908146034926176, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9953040480613708, "rewards/repeat_soft/std": 0.003897000104188919, "rewards/judge_quality/mean": 0.7675000429153442, "rewards/judge_quality/std": 0.20665708184242249, "rewards/total_composite/mean": 0.9235529899597168, "rewards/total_composite/std": 0.06322237104177475, "reward": 0.9235529899597168, "reward_std": 0.06322237104177475, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08199943602085114, "sampling/sampling_logp_difference/max": 2.096421003341675, "sampling/importance_sampling_ratio/min": 0.1228954866528511, "sampling/importance_sampling_ratio/mean": 1.0113463401794434, "sampling/importance_sampling_ratio/max": 1.7994391918182373, "entropy": 0.45118197798728943, "clip_ratio/low_mean": 0.028660714626312256, "clip_ratio/low_min": 0.028660714626312256, "clip_ratio/high_mean": 0.0474411235190928, "clip_ratio/high_max": 0.0474411235190928, "clip_ratio/region_mean": 0.07610183814540505, "reward_total_mean": 0.9235529899597168, "reward_meter_mean": 0.9861613512039185, "reward_meter_std": 0.005908146034926176, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9953040480613708, "reward_repeat_soft_std": 0.003897000104188919, "reward_judge_quality_mean": 0.7675000429153442, "reward_judge_quality_std": 0.20665708184242249, "reward_total_composite_mean": 0.9235529899597168, "reward_total_composite_std": 0.06322237104177475} {"timestamp_utc": "2026-04-13T03:53:23Z", "mode": "train", "global_step": 2217, "epoch": 0.2227021597187343, "loss": 0.0522, "grad_norm": 13.100147247314453, "learning_rate": 3.284848484848485e-06, "num_tokens": 4074907.0, "completions/mean_length": 43.0, "completions/min_length": 40.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.0, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.9392372369766235, "rewards/meter/std": 0.09251577407121658, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9840964674949646, "rewards/repeat_soft/std": 0.02281961962580681, "rewards/judge_quality/mean": 0.5924999713897705, "rewards/judge_quality/std": 0.2128547579050064, "rewards/total_composite/mean": 0.7511708736419678, "rewards/total_composite/std": 0.31035682559013367, "reward": 0.7511708736419678, "reward_std": 0.3103567957878113, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15737171471118927, "sampling/sampling_logp_difference/max": 2.8618855476379395, "sampling/importance_sampling_ratio/min": 0.0571608766913414, "sampling/importance_sampling_ratio/mean": 1.0188298225402832, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9522049203515053, "clip_ratio/low_mean": 0.011111111380159855, "clip_ratio/low_min": 0.011111111380159855, "clip_ratio/high_mean": 0.11994331050664186, "clip_ratio/high_max": 0.11994331050664186, "clip_ratio/region_mean": 0.13105442188680172, "reward_total_mean": 0.7511708736419678, "reward_meter_mean": 0.9392372369766235, "reward_meter_std": 0.09251577407121658, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9840964674949646, "reward_repeat_soft_std": 0.02281961962580681, "reward_judge_quality_mean": 0.5924999713897705, "reward_judge_quality_std": 0.2128547579050064, "reward_total_composite_mean": 0.7511708736419678, "reward_total_composite_std": 0.31035682559013367} {"timestamp_utc": "2026-04-13T03:53:29Z", "mode": "train", "global_step": 2218, "epoch": 0.22280261175288799, "loss": 0.1079, "grad_norm": 18.16419219970703, "learning_rate": 3.281818181818182e-06, "num_tokens": 4076226.0, "completions/mean_length": 27.875, "completions/min_length": 21.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.875, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.508415162563324, "rewards/meter/std": 0.4719443917274475, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.39249998331069946, "rewards/judge_quality/std": 0.08892211318016052, "rewards/total_composite/mean": 0.5927867889404297, "rewards/total_composite/std": 0.21094156801700592, "reward": 0.5927867889404297, "reward_std": 0.21094155311584473, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11778513342142105, "sampling/sampling_logp_difference/max": 1.3745298385620117, "sampling/importance_sampling_ratio/min": 0.2529584765434265, "sampling/importance_sampling_ratio/mean": 1.0319993495941162, "sampling/importance_sampling_ratio/max": 1.8838363885879517, "entropy": 0.938125304877758, "clip_ratio/low_mean": 0.08629186823964119, "clip_ratio/low_min": 0.08629186823964119, "clip_ratio/high_mean": 0.04869047645479441, "clip_ratio/high_max": 0.04869047645479441, "clip_ratio/region_mean": 0.1349823446944356, "reward_total_mean": 0.5927867889404297, "reward_meter_mean": 0.508415162563324, "reward_meter_std": 0.4719443917274475, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.39249998331069946, "reward_judge_quality_std": 0.08892211318016052, "reward_total_composite_mean": 0.5927867889404297, "reward_total_composite_std": 0.21094156801700592} {"timestamp_utc": "2026-04-13T03:53:36Z", "mode": "train", "global_step": 2219, "epoch": 0.2229030637870417, "loss": 0.0136, "grad_norm": 7.921211242675781, "learning_rate": 3.2787878787878793e-06, "num_tokens": 4078616.0, "completions/mean_length": 116.75, "completions/min_length": 103.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.75, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.8838856220245361, "rewards/meter/std": 0.08036492019891739, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9433990120887756, "rewards/repeat_soft/std": 0.025423500686883926, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7317134141921997, "rewards/total_composite/std": 0.03498329967260361, "reward": 0.7317134141921997, "reward_std": 0.0349833108484745, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11073584109544754, "sampling/sampling_logp_difference/max": 2.439612865447998, "sampling/importance_sampling_ratio/min": 0.08719459921121597, "sampling/importance_sampling_ratio/mean": 1.004769206047058, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47090277075767517, "clip_ratio/low_mean": 0.043316702358424664, "clip_ratio/low_min": 0.043316702358424664, "clip_ratio/high_mean": 0.04472707491368055, "clip_ratio/high_max": 0.04472707491368055, "clip_ratio/region_mean": 0.08804377727210522, "reward_total_mean": 0.7317134141921997, "reward_meter_mean": 0.8838856220245361, "reward_meter_std": 0.08036492019891739, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9433990120887756, "reward_repeat_soft_std": 0.025423500686883926, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7317134141921997, "reward_total_composite_std": 0.03498329967260361} {"timestamp_utc": "2026-04-13T03:53:48Z", "mode": "train", "global_step": 2220, "epoch": 0.22300351582119537, "loss": -0.1485, "grad_norm": 2.0679495334625244, "learning_rate": 3.275757575757576e-06, "num_tokens": 4080388.0, "completions/mean_length": 367.5, "completions/min_length": 123.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.625, "completions/mean_terminated_length": 126.66667175292969, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.7855796217918396, "rewards/meter/std": 0.2723711133003235, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.29880714416503906, "rewards/hard_gate/mean": 0.375, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9877635836601257, "rewards/repeat_soft/std": 0.012120707891881466, "rewards/judge_quality/mean": 0.18250000476837158, "rewards/judge_quality/std": 0.1973937302827835, "rewards/total_composite/mean": 0.2668796479701996, "rewards/total_composite/std": 0.37769100069999695, "reward": 0.2668796479701996, "reward_std": 0.37769100069999695, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14937126636505127, "sampling/sampling_logp_difference/max": 3.662163257598877, "sampling/importance_sampling_ratio/min": 0.02567690797150135, "sampling/importance_sampling_ratio/mean": 1.0147968530654907, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4093577340245247, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.046624258160591125, "clip_ratio/high_max": 0.046624258160591125, "clip_ratio/region_mean": 0.046624258160591125, "reward_total_mean": 0.2668796479701996, "reward_meter_mean": 0.7855796217918396, "reward_meter_std": 0.2723711133003235, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.29880714416503906, "reward_hard_gate_mean": 0.375, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9877635836601257, "reward_repeat_soft_std": 0.012120707891881466, "reward_judge_quality_mean": 0.18250000476837158, "reward_judge_quality_std": 0.1973937302827835, "reward_total_composite_mean": 0.2668796479701996, "reward_total_composite_std": 0.37769100069999695} {"timestamp_utc": "2026-04-13T03:53:55Z", "mode": "train", "global_step": 2221, "epoch": 0.22310396785534908, "loss": 0.0437, "grad_norm": 7.859836101531982, "learning_rate": 3.272727272727273e-06, "num_tokens": 4082450.0, "completions/mean_length": 86.75, "completions/min_length": 74.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.75, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.7221540808677673, "rewards/meter/std": 0.2535248100757599, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9498318433761597, "rewards/repeat_soft/std": 0.0456172414124012, "rewards/judge_quality/mean": 0.4737499952316284, "rewards/judge_quality/std": 0.16291432082653046, "rewards/total_composite/mean": 0.702702522277832, "rewards/total_composite/std": 0.11635494232177734, "reward": 0.702702522277832, "reward_std": 0.11635493487119675, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1250755786895752, "sampling/sampling_logp_difference/max": 2.619292974472046, "sampling/importance_sampling_ratio/min": 0.07285435497760773, "sampling/importance_sampling_ratio/mean": 0.9980260729789734, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6256414465606213, "clip_ratio/low_mean": 0.04997573234140873, "clip_ratio/low_min": 0.04997573234140873, "clip_ratio/high_mean": 0.06543326657265425, "clip_ratio/high_max": 0.06543326657265425, "clip_ratio/region_mean": 0.11540899891406298, "reward_total_mean": 0.702702522277832, "reward_meter_mean": 0.7221540808677673, "reward_meter_std": 0.2535248100757599, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9498318433761597, "reward_repeat_soft_std": 0.0456172414124012, "reward_judge_quality_mean": 0.4737499952316284, "reward_judge_quality_std": 0.16291432082653046, "reward_total_composite_mean": 0.702702522277832, "reward_total_composite_std": 0.11635494232177734} {"timestamp_utc": "2026-04-13T03:54:02Z", "mode": "train", "global_step": 2222, "epoch": 0.22320441988950276, "loss": 0.0199, "grad_norm": 10.231766700744629, "learning_rate": 3.2696969696969698e-06, "num_tokens": 4084104.0, "completions/mean_length": 53.75, "completions/min_length": 47.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.75, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.981306254863739, "rewards/meter/std": 0.015285252593457699, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9685190320014954, "rewards/repeat_soft/std": 0.023936370387673378, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8178147077560425, "rewards/total_composite/std": 0.008007273077964783, "reward": 0.8178147077560425, "reward_std": 0.008007258176803589, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12285489588975906, "sampling/sampling_logp_difference/max": 1.0942583084106445, "sampling/importance_sampling_ratio/min": 0.3347878158092499, "sampling/importance_sampling_ratio/mean": 1.0193394422531128, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7527876272797585, "clip_ratio/low_mean": 0.0562098752707243, "clip_ratio/low_min": 0.0562098752707243, "clip_ratio/high_mean": 0.09016973059624434, "clip_ratio/high_max": 0.09016973059624434, "clip_ratio/region_mean": 0.14637960586696863, "reward_total_mean": 0.8178147077560425, "reward_meter_mean": 0.981306254863739, "reward_meter_std": 0.015285252593457699, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9685190320014954, "reward_repeat_soft_std": 0.023936370387673378, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8178147077560425, "reward_total_composite_std": 0.008007273077964783} {"timestamp_utc": "2026-04-13T03:54:14Z", "mode": "train", "global_step": 2223, "epoch": 0.22330487192365644, "loss": -0.1408, "grad_norm": 2.280388355255127, "learning_rate": 3.266666666666667e-06, "num_tokens": 4085779.0, "completions/mean_length": 112.375, "completions/min_length": 48.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 55.28571701049805, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.7980515360832214, "rewards/meter/std": 0.3593797981739044, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9738426208496094, "rewards/repeat_soft/std": 0.034654486924409866, "rewards/judge_quality/mean": 0.6800000071525574, "rewards/judge_quality/std": 0.3556884825229645, "rewards/total_composite/mean": 0.7717449069023132, "rewards/total_composite/std": 0.3311907649040222, "reward": 0.7717449069023132, "reward_std": 0.3311907649040222, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10749486088752747, "sampling/sampling_logp_difference/max": 1.0331215858459473, "sampling/importance_sampling_ratio/min": 0.3558942973613739, "sampling/importance_sampling_ratio/mean": 1.0076767206192017, "sampling/importance_sampling_ratio/max": 1.6287171840667725, "entropy": 0.6088745556771755, "clip_ratio/low_mean": 0.023148147389292717, "clip_ratio/low_min": 0.023148147389292717, "clip_ratio/high_mean": 0.07147680874913931, "clip_ratio/high_max": 0.07147680874913931, "clip_ratio/region_mean": 0.09462495613843203, "reward_total_mean": 0.7717449069023132, "reward_meter_mean": 0.7980515360832214, "reward_meter_std": 0.3593797981739044, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9738426208496094, "reward_repeat_soft_std": 0.034654486924409866, "reward_judge_quality_mean": 0.6800000071525574, "reward_judge_quality_std": 0.3556884825229645, "reward_total_composite_mean": 0.7717449069023132, "reward_total_composite_std": 0.3311907649040222} {"timestamp_utc": "2026-04-13T03:54:20Z", "mode": "train", "global_step": 2224, "epoch": 0.22340532395781015, "loss": 0.0383, "grad_norm": 13.20445728302002, "learning_rate": 3.263636363636364e-06, "num_tokens": 4087357.0, "completions/mean_length": 31.25, "completions/min_length": 29.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.25, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.8193243741989136, "rewards/meter/std": 0.3482295274734497, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9527064561843872, "rewards/repeat_soft/std": 0.0166556928306818, "rewards/judge_quality/mean": 0.6399999856948853, "rewards/judge_quality/std": 0.3131864070892334, "rewards/total_composite/mean": 0.8059666156768799, "rewards/total_composite/std": 0.2184518724679947, "reward": 0.8059666156768799, "reward_std": 0.2184518426656723, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15950818359851837, "sampling/sampling_logp_difference/max": 1.7097837924957275, "sampling/importance_sampling_ratio/min": 0.18090490996837616, "sampling/importance_sampling_ratio/mean": 1.0016908645629883, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7782374173402786, "clip_ratio/low_mean": 0.014705882407724857, "clip_ratio/low_min": 0.014705882407724857, "clip_ratio/high_mean": 0.1292632259428501, "clip_ratio/high_max": 0.1292632259428501, "clip_ratio/region_mean": 0.14396910835057497, "reward_total_mean": 0.8059666156768799, "reward_meter_mean": 0.8193243741989136, "reward_meter_std": 0.3482295274734497, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9527064561843872, "reward_repeat_soft_std": 0.0166556928306818, "reward_judge_quality_mean": 0.6399999856948853, "reward_judge_quality_std": 0.3131864070892334, "reward_total_composite_mean": 0.8059666156768799, "reward_total_composite_std": 0.2184518724679947} {"timestamp_utc": "2026-04-13T03:54:31Z", "mode": "train", "global_step": 2225, "epoch": 0.22350577599196383, "loss": -0.1543, "grad_norm": 1.6287188529968262, "learning_rate": 3.2606060606060607e-06, "num_tokens": 4089134.0, "completions/mean_length": 116.125, "completions/min_length": 58.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 59.57143020629883, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.889602780342102, "rewards/meter/std": 0.258131742477417, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9015271067619324, "rewards/repeat_soft/std": 0.05359543114900589, "rewards/judge_quality/mean": 0.4150000214576721, "rewards/judge_quality/std": 0.18031719326972961, "rewards/total_composite/mean": 0.7185272574424744, "rewards/total_composite/std": 0.29200807213783264, "reward": 0.7185272574424744, "reward_std": 0.29200804233551025, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10081452876329422, "sampling/sampling_logp_difference/max": 1.7857608795166016, "sampling/importance_sampling_ratio/min": 0.16766944527626038, "sampling/importance_sampling_ratio/mean": 1.0050296783447266, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.536356620490551, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10293604130856693, "clip_ratio/high_max": 0.10293604130856693, "clip_ratio/region_mean": 0.10293604130856693, "reward_total_mean": 0.7185272574424744, "reward_meter_mean": 0.889602780342102, "reward_meter_std": 0.258131742477417, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9015271067619324, "reward_repeat_soft_std": 0.05359543114900589, "reward_judge_quality_mean": 0.4150000214576721, "reward_judge_quality_std": 0.18031719326972961, "reward_total_composite_mean": 0.7185272574424744, "reward_total_composite_std": 0.29200807213783264} {"timestamp_utc": "2026-04-13T03:54:37Z", "mode": "train", "global_step": 2226, "epoch": 0.22360622802611752, "loss": 0.1031, "grad_norm": 20.077205657958984, "learning_rate": 3.257575757575758e-06, "num_tokens": 4090687.0, "completions/mean_length": 40.125, "completions/min_length": 35.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.125, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.8581901788711548, "rewards/meter/std": 0.24343305826187134, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.986177921295166, "rewards/repeat_soft/std": 0.018086345866322517, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7630533576011658, "rewards/total_composite/std": 0.11021636426448822, "reward": 0.7630533576011658, "reward_std": 0.11021635681390762, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1063460111618042, "sampling/sampling_logp_difference/max": 1.4289612770080566, "sampling/importance_sampling_ratio/min": 0.23955762386322021, "sampling/importance_sampling_ratio/mean": 1.0207276344299316, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6209010295569897, "clip_ratio/low_mean": 0.01595744676887989, "clip_ratio/low_min": 0.01595744676887989, "clip_ratio/high_mean": 0.07642945973202586, "clip_ratio/high_max": 0.07642945973202586, "clip_ratio/region_mean": 0.09238690650090575, "reward_total_mean": 0.7630533576011658, "reward_meter_mean": 0.8581901788711548, "reward_meter_std": 0.24343305826187134, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.986177921295166, "reward_repeat_soft_std": 0.018086345866322517, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7630533576011658, "reward_total_composite_std": 0.11021636426448822} {"timestamp_utc": "2026-04-13T03:54:49Z", "mode": "train", "global_step": 2227, "epoch": 0.22370668006027122, "loss": -0.1665, "grad_norm": 3.0309407711029053, "learning_rate": 3.2545454545454548e-06, "num_tokens": 4092768.0, "completions/mean_length": 151.125, "completions/min_length": 87.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 99.5714340209961, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.47105085849761963, "rewards/meter/std": 0.32570669054985046, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9498853087425232, "rewards/repeat_soft/std": 0.06039560213685036, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.11686377227306366, "rewards/total_composite/mean": 0.5226582288742065, "rewards/total_composite/std": 0.2387293577194214, "reward": 0.5226582288742065, "reward_std": 0.238729327917099, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14574559032917023, "sampling/sampling_logp_difference/max": 1.8899831771850586, "sampling/importance_sampling_ratio/min": 0.1510743498802185, "sampling/importance_sampling_ratio/mean": 1.0035481452941895, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7914652302861214, "clip_ratio/low_mean": 0.018041236326098442, "clip_ratio/low_min": 0.018041236326098442, "clip_ratio/high_mean": 0.09371810173615813, "clip_ratio/high_max": 0.09371810173615813, "clip_ratio/region_mean": 0.11175933806225657, "reward_total_mean": 0.5226582288742065, "reward_meter_mean": 0.47105085849761963, "reward_meter_std": 0.32570669054985046, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9498853087425232, "reward_repeat_soft_std": 0.06039560213685036, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.11686377227306366, "reward_total_composite_mean": 0.5226582288742065, "reward_total_composite_std": 0.2387293577194214} {"timestamp_utc": "2026-04-13T03:54:56Z", "mode": "train", "global_step": 2228, "epoch": 0.2238071320944249, "loss": 0.0508, "grad_norm": 8.015661239624023, "learning_rate": 3.2515151515151516e-06, "num_tokens": 4094891.0, "completions/mean_length": 91.375, "completions/min_length": 75.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.375, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.9621787667274475, "rewards/meter/std": 0.06725805997848511, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9921069145202637, "rewards/repeat_soft/std": 0.011120432056486607, "rewards/judge_quality/mean": 0.6649999618530273, "rewards/judge_quality/std": 0.2622430920600891, "rewards/total_composite/mean": 0.8754411339759827, "rewards/total_composite/std": 0.0737392008304596, "reward": 0.8754411339759827, "reward_std": 0.0737392008304596, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11661697924137115, "sampling/sampling_logp_difference/max": 1.336998462677002, "sampling/importance_sampling_ratio/min": 0.2626327872276306, "sampling/importance_sampling_ratio/mean": 1.0102447271347046, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7690745964646339, "clip_ratio/low_mean": 0.06149100139737129, "clip_ratio/low_min": 0.06149100139737129, "clip_ratio/high_mean": 0.04253565240651369, "clip_ratio/high_max": 0.04253565240651369, "clip_ratio/region_mean": 0.10402665380388498, "reward_total_mean": 0.8754411339759827, "reward_meter_mean": 0.9621787667274475, "reward_meter_std": 0.06725805997848511, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9921069145202637, "reward_repeat_soft_std": 0.011120432056486607, "reward_judge_quality_mean": 0.6649999618530273, "reward_judge_quality_std": 0.2622430920600891, "reward_total_composite_mean": 0.8754411339759827, "reward_total_composite_std": 0.0737392008304596} {"timestamp_utc": "2026-04-13T03:55:08Z", "mode": "train", "global_step": 2229, "epoch": 0.2239075841285786, "loss": -0.1648, "grad_norm": 2.688572406768799, "learning_rate": 3.2484848484848484e-06, "num_tokens": 4096960.0, "completions/mean_length": 145.625, "completions/min_length": 88.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 93.28572082519531, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.4742259979248047, "rewards/meter/std": 0.2217399626970291, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9784789085388184, "rewards/repeat_soft/std": 0.017728574573993683, "rewards/judge_quality/mean": 0.44874998927116394, "rewards/judge_quality/std": 0.2105392962694168, "rewards/total_composite/mean": 0.5483515858650208, "rewards/total_composite/std": 0.2440459430217743, "reward": 0.5483515858650208, "reward_std": 0.2440459430217743, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1098293587565422, "sampling/sampling_logp_difference/max": 1.5664641857147217, "sampling/importance_sampling_ratio/min": 0.20878209173679352, "sampling/importance_sampling_ratio/mean": 1.007502555847168, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4733167551457882, "clip_ratio/low_mean": 0.02730414690449834, "clip_ratio/low_min": 0.02730414690449834, "clip_ratio/high_mean": 0.07723797857761383, "clip_ratio/high_max": 0.07723797857761383, "clip_ratio/region_mean": 0.10454212548211217, "reward_total_mean": 0.5483515858650208, "reward_meter_mean": 0.4742259979248047, "reward_meter_std": 0.2217399626970291, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9784789085388184, "reward_repeat_soft_std": 0.017728574573993683, "reward_judge_quality_mean": 0.44874998927116394, "reward_judge_quality_std": 0.2105392962694168, "reward_total_composite_mean": 0.5483515858650208, "reward_total_composite_std": 0.2440459430217743} {"timestamp_utc": "2026-04-13T03:55:15Z", "mode": "train", "global_step": 2230, "epoch": 0.2240080361627323, "loss": 0.0378, "grad_norm": 8.90649127960205, "learning_rate": 3.2454545454545457e-06, "num_tokens": 4098544.0, "completions/mean_length": 51.0, "completions/min_length": 44.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.0, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.7260185480117798, "rewards/meter/std": 0.2914537787437439, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9909524321556091, "rewards/repeat_soft/std": 0.010254411958158016, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404788017273, "rewards/total_composite/mean": 0.7426785826683044, "rewards/total_composite/std": 0.09957020729780197, "reward": 0.7426785826683044, "reward_std": 0.09957019984722137, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11121776700019836, "sampling/sampling_logp_difference/max": 1.2877588272094727, "sampling/importance_sampling_ratio/min": 0.2758884131908417, "sampling/importance_sampling_ratio/mean": 0.9991315603256226, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6334501430392265, "clip_ratio/low_mean": 0.039941984228789806, "clip_ratio/low_min": 0.039941984228789806, "clip_ratio/high_mean": 0.06984772719442844, "clip_ratio/high_max": 0.06984772719442844, "clip_ratio/region_mean": 0.10978971142321825, "reward_total_mean": 0.7426785826683044, "reward_meter_mean": 0.7260185480117798, "reward_meter_std": 0.2914537787437439, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9909524321556091, "reward_repeat_soft_std": 0.010254411958158016, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404788017273, "reward_total_composite_mean": 0.7426785826683044, "reward_total_composite_std": 0.09957020729780197} {"timestamp_utc": "2026-04-13T03:55:22Z", "mode": "train", "global_step": 2231, "epoch": 0.22410848819688597, "loss": 0.1392, "grad_norm": 8.986328125, "learning_rate": 3.2424242424242425e-06, "num_tokens": 4100636.0, "completions/mean_length": 90.5, "completions/min_length": 61.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.5, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.9107909202575684, "rewards/meter/std": 0.15899308025836945, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9413553476333618, "rewards/repeat_soft/std": 0.042977601289749146, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.7349914312362671, "rewards/total_composite/std": 0.0590631328523159, "reward": 0.7349914312362671, "reward_std": 0.05906311422586441, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12965568900108337, "sampling/sampling_logp_difference/max": 2.645667791366577, "sampling/importance_sampling_ratio/min": 0.0709579586982727, "sampling/importance_sampling_ratio/mean": 1.0074139833450317, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.78621556609869, "clip_ratio/low_mean": 0.04109727684408426, "clip_ratio/low_min": 0.04109727684408426, "clip_ratio/high_mean": 0.05750239174813032, "clip_ratio/high_max": 0.05750239174813032, "clip_ratio/region_mean": 0.09859966859221458, "reward_total_mean": 0.7349914312362671, "reward_meter_mean": 0.9107909202575684, "reward_meter_std": 0.15899308025836945, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9413553476333618, "reward_repeat_soft_std": 0.042977601289749146, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.7349914312362671, "reward_total_composite_std": 0.0590631328523159} {"timestamp_utc": "2026-04-13T03:55:29Z", "mode": "train", "global_step": 2232, "epoch": 0.22420894023103968, "loss": 0.0741, "grad_norm": 10.511190414428711, "learning_rate": 3.2393939393939393e-06, "num_tokens": 4102703.0, "completions/mean_length": 95.375, "completions/min_length": 82.0, "completions/max_length": 113.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.375, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.9880343675613403, "rewards/meter/std": 0.003099605208262801, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9585582613945007, "rewards/repeat_soft/std": 0.02052144519984722, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.8054088354110718, "rewards/total_composite/std": 0.02137584052979946, "reward": 0.8054088354110718, "reward_std": 0.021375831216573715, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10798489302396774, "sampling/sampling_logp_difference/max": 1.3944332599639893, "sampling/importance_sampling_ratio/min": 0.24797353148460388, "sampling/importance_sampling_ratio/mean": 1.0042060613632202, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5419639497995377, "clip_ratio/low_mean": 0.01716348435729742, "clip_ratio/low_min": 0.01716348435729742, "clip_ratio/high_mean": 0.05579789215698838, "clip_ratio/high_max": 0.05579789215698838, "clip_ratio/region_mean": 0.0729613765142858, "reward_total_mean": 0.8054088354110718, "reward_meter_mean": 0.9880343675613403, "reward_meter_std": 0.003099605208262801, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9585582613945007, "reward_repeat_soft_std": 0.02052144519984722, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.8054088354110718, "reward_total_composite_std": 0.02137584052979946} {"timestamp_utc": "2026-04-13T03:55:36Z", "mode": "train", "global_step": 2233, "epoch": 0.22430939226519336, "loss": 0.0067, "grad_norm": 9.848464965820312, "learning_rate": 3.236363636363636e-06, "num_tokens": 4104395.0, "completions/mean_length": 53.5, "completions/min_length": 49.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.5, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9595493674278259, "rewards/meter/std": 0.034741006791591644, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9810022115707397, "rewards/repeat_soft/std": 0.020748134702444077, "rewards/judge_quality/mean": 0.5049999952316284, "rewards/judge_quality/std": 0.16801361739635468, "rewards/total_composite/mean": 0.8313974142074585, "rewards/total_composite/std": 0.05657944828271866, "reward": 0.8313974142074585, "reward_std": 0.05657944455742836, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12310612201690674, "sampling/sampling_logp_difference/max": 1.1195611953735352, "sampling/importance_sampling_ratio/min": 0.3264230191707611, "sampling/importance_sampling_ratio/mean": 1.006029486656189, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6894506439566612, "clip_ratio/low_mean": 0.09705463610589504, "clip_ratio/low_min": 0.09705463610589504, "clip_ratio/high_mean": 0.01116071455180645, "clip_ratio/high_max": 0.01116071455180645, "clip_ratio/region_mean": 0.10821535065770149, "reward_total_mean": 0.8313974142074585, "reward_meter_mean": 0.9595493674278259, "reward_meter_std": 0.034741006791591644, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9810022115707397, "reward_repeat_soft_std": 0.020748134702444077, "reward_judge_quality_mean": 0.5049999952316284, "reward_judge_quality_std": 0.16801361739635468, "reward_total_composite_mean": 0.8313974142074585, "reward_total_composite_std": 0.05657944828271866} {"timestamp_utc": "2026-04-13T03:55:42Z", "mode": "train", "global_step": 2234, "epoch": 0.22440984429934707, "loss": 0.0396, "grad_norm": 10.070225715637207, "learning_rate": 3.2333333333333334e-06, "num_tokens": 4106240.0, "completions/mean_length": 55.625, "completions/min_length": 52.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.625, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9863675832748413, "rewards/meter/std": 0.006686607841402292, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8766511678695679, "rewards/repeat_soft/std": 0.09200531244277954, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8097805380821228, "rewards/total_composite/std": 0.012743495404720306, "reward": 0.8097805380821228, "reward_std": 0.012743494473397732, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07505901157855988, "sampling/sampling_logp_difference/max": 1.5159797668457031, "sampling/importance_sampling_ratio/min": 0.21959294378757477, "sampling/importance_sampling_ratio/mean": 1.0122835636138916, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4108808785676956, "clip_ratio/low_mean": 0.03091056109406054, "clip_ratio/low_min": 0.03091056109406054, "clip_ratio/high_mean": 0.03635965380817652, "clip_ratio/high_max": 0.03635965380817652, "clip_ratio/region_mean": 0.06727021490223706, "reward_total_mean": 0.8097805380821228, "reward_meter_mean": 0.9863675832748413, "reward_meter_std": 0.006686607841402292, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8766511678695679, "reward_repeat_soft_std": 0.09200531244277954, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8097805380821228, "reward_total_composite_std": 0.012743495404720306} {"timestamp_utc": "2026-04-13T03:55:49Z", "mode": "train", "global_step": 2235, "epoch": 0.22451029633350075, "loss": 0.1655, "grad_norm": 23.532915115356445, "learning_rate": 3.2303030303030307e-06, "num_tokens": 4107673.0, "completions/mean_length": 33.125, "completions/min_length": 28.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.125, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.7942683100700378, "rewards/meter/std": 0.3125913739204407, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9499238133430481, "rewards/repeat_soft/std": 0.040320299565792084, "rewards/judge_quality/mean": 0.35999998450279236, "rewards/judge_quality/std": 0.09165150672197342, "rewards/total_composite/mean": 0.7104130983352661, "rewards/total_composite/std": 0.1488594114780426, "reward": 0.7104130983352661, "reward_std": 0.1488594114780426, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1119021624326706, "sampling/sampling_logp_difference/max": 1.5682306289672852, "sampling/importance_sampling_ratio/min": 0.3441673517227173, "sampling/importance_sampling_ratio/mean": 1.015417218208313, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6441926807165146, "clip_ratio/low_mean": 0.026863354723900557, "clip_ratio/low_min": 0.026863354723900557, "clip_ratio/high_mean": 0.11854171566665173, "clip_ratio/high_max": 0.11854171566665173, "clip_ratio/region_mean": 0.14540507039055228, "reward_total_mean": 0.7104130983352661, "reward_meter_mean": 0.7942683100700378, "reward_meter_std": 0.3125913739204407, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9499238133430481, "reward_repeat_soft_std": 0.040320299565792084, "reward_judge_quality_mean": 0.35999998450279236, "reward_judge_quality_std": 0.09165150672197342, "reward_total_composite_mean": 0.7104130983352661, "reward_total_composite_std": 0.1488594114780426} {"timestamp_utc": "2026-04-13T03:55:56Z", "mode": "train", "global_step": 2236, "epoch": 0.22461074836765443, "loss": 0.1145, "grad_norm": 10.464608192443848, "learning_rate": 3.227272727272728e-06, "num_tokens": 4109649.0, "completions/mean_length": 79.0, "completions/min_length": 58.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.8168066143989563, "rewards/meter/std": 0.19019253551959991, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.26726123690605164, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9541059136390686, "rewards/repeat_soft/std": 0.034865520894527435, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.6950985789299011, "rewards/total_composite/std": 0.10532040894031525, "reward": 0.6950985789299011, "reward_std": 0.10532040894031525, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11661787331104279, "sampling/sampling_logp_difference/max": 1.8038818836212158, "sampling/importance_sampling_ratio/min": 0.18611937761306763, "sampling/importance_sampling_ratio/mean": 1.0172001123428345, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.629978321492672, "clip_ratio/low_mean": 0.046602402813732624, "clip_ratio/low_min": 0.046602402813732624, "clip_ratio/high_mean": 0.060104165226221085, "clip_ratio/high_max": 0.060104165226221085, "clip_ratio/region_mean": 0.10670656803995371, "reward_total_mean": 0.6950985789299011, "reward_meter_mean": 0.8168066143989563, "reward_meter_std": 0.19019253551959991, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.26726123690605164, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9541059136390686, "reward_repeat_soft_std": 0.034865520894527435, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.6950985789299011, "reward_total_composite_std": 0.10532040894031525} {"timestamp_utc": "2026-04-13T03:56:03Z", "mode": "train", "global_step": 2237, "epoch": 0.22471120040180814, "loss": 0.0635, "grad_norm": 10.85342025756836, "learning_rate": 3.2242424242424248e-06, "num_tokens": 4111545.0, "completions/mean_length": 76.0, "completions/min_length": 62.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.8412046432495117, "rewards/meter/std": 0.11123973876237869, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9913831949234009, "rewards/repeat_soft/std": 0.00919759925454855, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.784930408000946, "rewards/total_composite/std": 0.0878324806690216, "reward": 0.784930408000946, "reward_std": 0.08783247321844101, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10584114491939545, "sampling/sampling_logp_difference/max": 1.3403987884521484, "sampling/importance_sampling_ratio/min": 0.2617412805557251, "sampling/importance_sampling_ratio/mean": 1.0052521228790283, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6661771908402443, "clip_ratio/low_mean": 0.05632713157683611, "clip_ratio/low_min": 0.05632713157683611, "clip_ratio/high_mean": 0.04631362482905388, "clip_ratio/high_max": 0.04631362482905388, "clip_ratio/region_mean": 0.10264075640588999, "reward_total_mean": 0.784930408000946, "reward_meter_mean": 0.8412046432495117, "reward_meter_std": 0.11123973876237869, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9913831949234009, "reward_repeat_soft_std": 0.00919759925454855, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.784930408000946, "reward_total_composite_std": 0.0878324806690216} {"timestamp_utc": "2026-04-13T03:56:10Z", "mode": "train", "global_step": 2238, "epoch": 0.22481165243596182, "loss": 0.0055, "grad_norm": 5.718544006347656, "learning_rate": 3.2212121212121216e-06, "num_tokens": 4113654.0, "completions/mean_length": 104.625, "completions/min_length": 94.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.625, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.9781730771064758, "rewards/meter/std": 0.01380665972828865, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9009831547737122, "rewards/repeat_soft/std": 0.10116281360387802, "rewards/judge_quality/mean": 0.30124998092651367, "rewards/judge_quality/std": 0.10398317128419876, "rewards/total_composite/mean": 0.7706512212753296, "rewards/total_composite/std": 0.03468724712729454, "reward": 0.7706512212753296, "reward_std": 0.03468725457787514, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0829453319311142, "sampling/sampling_logp_difference/max": 1.4270105361938477, "sampling/importance_sampling_ratio/min": 0.24002541601657867, "sampling/importance_sampling_ratio/mean": 1.0023982524871826, "sampling/importance_sampling_ratio/max": 1.8992918729782104, "entropy": 0.45193910226225853, "clip_ratio/low_mean": 0.049505599308758974, "clip_ratio/low_min": 0.049505599308758974, "clip_ratio/high_mean": 0.04791781306266785, "clip_ratio/high_max": 0.04791781306266785, "clip_ratio/region_mean": 0.09742341237142682, "reward_total_mean": 0.7706512212753296, "reward_meter_mean": 0.9781730771064758, "reward_meter_std": 0.01380665972828865, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9009831547737122, "reward_repeat_soft_std": 0.10116281360387802, "reward_judge_quality_mean": 0.30124998092651367, "reward_judge_quality_std": 0.10398317128419876, "reward_total_composite_mean": 0.7706512212753296, "reward_total_composite_std": 0.03468724712729454} {"timestamp_utc": "2026-04-13T03:56:17Z", "mode": "train", "global_step": 2239, "epoch": 0.22491210447011553, "loss": 0.032, "grad_norm": 9.57489013671875, "learning_rate": 3.2181818181818184e-06, "num_tokens": 4115823.0, "completions/mean_length": 90.125, "completions/min_length": 77.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.125, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.8767380714416504, "rewards/meter/std": 0.23509320616722107, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.978999137878418, "rewards/repeat_soft/std": 0.01253342442214489, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7430570125579834, "rewards/total_composite/std": 0.09886626899242401, "reward": 0.7430570125579834, "reward_std": 0.09886626899242401, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13204294443130493, "sampling/sampling_logp_difference/max": 1.6245622634887695, "sampling/importance_sampling_ratio/min": 0.19699789583683014, "sampling/importance_sampling_ratio/mean": 1.009281873703003, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8193022683262825, "clip_ratio/low_mean": 0.034556809812784195, "clip_ratio/low_min": 0.034556809812784195, "clip_ratio/high_mean": 0.10091532208025455, "clip_ratio/high_max": 0.10091532208025455, "clip_ratio/region_mean": 0.13547213189303875, "reward_total_mean": 0.7430570125579834, "reward_meter_mean": 0.8767380714416504, "reward_meter_std": 0.23509320616722107, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.978999137878418, "reward_repeat_soft_std": 0.01253342442214489, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7430570125579834, "reward_total_composite_std": 0.09886626899242401} {"timestamp_utc": "2026-04-13T03:56:23Z", "mode": "train", "global_step": 2240, "epoch": 0.2250125565042692, "loss": 0.025, "grad_norm": 16.934492111206055, "learning_rate": 3.2151515151515157e-06, "num_tokens": 4117251.0, "completions/mean_length": 27.5, "completions/min_length": 24.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.5, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9328964948654175, "rewards/meter/std": 0.07221292704343796, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9585226774215698, "rewards/repeat_soft/std": 0.011249415576457977, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7916556596755981, "rewards/total_composite/std": 0.03220982104539871, "reward": 0.7916556596755981, "reward_std": 0.03220982849597931, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12131853401660919, "sampling/sampling_logp_difference/max": 2.9359490871429443, "sampling/importance_sampling_ratio/min": 0.05308031663298607, "sampling/importance_sampling_ratio/mean": 1.0058128833770752, "sampling/importance_sampling_ratio/max": 1.7095675468444824, "entropy": 0.7615194246172905, "clip_ratio/low_mean": 0.03212932962924242, "clip_ratio/low_min": 0.03212932962924242, "clip_ratio/high_mean": 0.06795566529035568, "clip_ratio/high_max": 0.06795566529035568, "clip_ratio/region_mean": 0.1000849949195981, "reward_total_mean": 0.7916556596755981, "reward_meter_mean": 0.9328964948654175, "reward_meter_std": 0.07221292704343796, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9585226774215698, "reward_repeat_soft_std": 0.011249415576457977, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7916556596755981, "reward_total_composite_std": 0.03220982104539871} {"timestamp_utc": "2026-04-13T03:56:30Z", "mode": "train", "global_step": 2241, "epoch": 0.2251130085384229, "loss": 0.0187, "grad_norm": 11.523473739624023, "learning_rate": 3.2121212121212125e-06, "num_tokens": 4118837.0, "completions/mean_length": 50.25, "completions/min_length": 40.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.25, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9816967248916626, "rewards/meter/std": 0.008510376326739788, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9702972173690796, "rewards/repeat_soft/std": 0.025310644879937172, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.8920432329177856, "rewards/total_composite/std": 0.07569780945777893, "reward": 0.8920432329177856, "reward_std": 0.07569781690835953, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12100984156131744, "sampling/sampling_logp_difference/max": 1.4770324230194092, "sampling/importance_sampling_ratio/min": 0.22831423580646515, "sampling/importance_sampling_ratio/mean": 1.0129563808441162, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6967472732067108, "clip_ratio/low_mean": 0.05686284601688385, "clip_ratio/low_min": 0.05686284601688385, "clip_ratio/high_mean": 0.06146885501220822, "clip_ratio/high_max": 0.06146885501220822, "clip_ratio/region_mean": 0.11833170102909207, "reward_total_mean": 0.8920432329177856, "reward_meter_mean": 0.9816967248916626, "reward_meter_std": 0.008510376326739788, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9702972173690796, "reward_repeat_soft_std": 0.025310644879937172, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.8920432329177856, "reward_total_composite_std": 0.07569780945777893} {"timestamp_utc": "2026-04-13T03:56:37Z", "mode": "train", "global_step": 2242, "epoch": 0.2252134605725766, "loss": 0.0234, "grad_norm": 8.19454288482666, "learning_rate": 3.2090909090909094e-06, "num_tokens": 4120891.0, "completions/mean_length": 96.75, "completions/min_length": 91.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.75, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.7492620944976807, "rewards/meter/std": 0.2832441031932831, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9640977382659912, "rewards/repeat_soft/std": 0.02658494934439659, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.6880152225494385, "rewards/total_composite/std": 0.1428159475326538, "reward": 0.6880152225494385, "reward_std": 0.142815962433815, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1078050509095192, "sampling/sampling_logp_difference/max": 1.6046545505523682, "sampling/importance_sampling_ratio/min": 0.2009589672088623, "sampling/importance_sampling_ratio/mean": 1.0091211795806885, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5461454093456268, "clip_ratio/low_mean": 0.031037233769893646, "clip_ratio/low_min": 0.031037233769893646, "clip_ratio/high_mean": 0.09156281687319279, "clip_ratio/high_max": 0.09156281687319279, "clip_ratio/region_mean": 0.12260005064308643, "reward_total_mean": 0.6880152225494385, "reward_meter_mean": 0.7492620944976807, "reward_meter_std": 0.2832441031932831, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9640977382659912, "reward_repeat_soft_std": 0.02658494934439659, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.6880152225494385, "reward_total_composite_std": 0.1428159475326538} {"timestamp_utc": "2026-04-13T03:56:46Z", "mode": "train", "global_step": 2243, "epoch": 0.22531391260673028, "loss": 0.0083, "grad_norm": 6.431366920471191, "learning_rate": 3.2060606060606066e-06, "num_tokens": 4123817.0, "completions/mean_length": 168.75, "completions/min_length": 146.0, "completions/max_length": 191.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 168.75, "completions/min_terminated_length": 146.0, "completions/max_terminated_length": 191.0, "rewards/meter/mean": 0.9668172597885132, "rewards/meter/std": 0.05546333268284798, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9325956106185913, "rewards/repeat_soft/std": 0.026578247547149658, "rewards/judge_quality/mean": 0.22499999403953552, "rewards/judge_quality/std": 0.04629100486636162, "rewards/total_composite/mean": 0.7383273839950562, "rewards/total_composite/std": 0.028311023488640785, "reward": 0.7383273839950562, "reward_std": 0.028311021625995636, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12992294132709503, "sampling/sampling_logp_difference/max": 2.1137452125549316, "sampling/importance_sampling_ratio/min": 0.21811451017856598, "sampling/importance_sampling_ratio/mean": 1.0178297758102417, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.022638663649559, "clip_ratio/low_mean": 0.03576062060892582, "clip_ratio/low_min": 0.03576062060892582, "clip_ratio/high_mean": 0.08282111119478941, "clip_ratio/high_max": 0.08282111119478941, "clip_ratio/region_mean": 0.11858173180371523, "reward_total_mean": 0.7383273839950562, "reward_meter_mean": 0.9668172597885132, "reward_meter_std": 0.05546333268284798, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9325956106185913, "reward_repeat_soft_std": 0.026578247547149658, "reward_judge_quality_mean": 0.22499999403953552, "reward_judge_quality_std": 0.04629100486636162, "reward_total_composite_mean": 0.7383273839950562, "reward_total_composite_std": 0.028311023488640785} {"timestamp_utc": "2026-04-13T03:56:54Z", "mode": "train", "global_step": 2244, "epoch": 0.225414364640884, "loss": 0.0063, "grad_norm": 6.013772964477539, "learning_rate": 3.2030303030303034e-06, "num_tokens": 4126733.0, "completions/mean_length": 163.5, "completions/min_length": 147.0, "completions/max_length": 176.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 163.5, "completions/min_terminated_length": 147.0, "completions/max_terminated_length": 176.0, "rewards/meter/mean": 0.9124640822410583, "rewards/meter/std": 0.07928428798913956, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9853527545928955, "rewards/repeat_soft/std": 0.01585215888917446, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7851440906524658, "rewards/total_composite/std": 0.03541453182697296, "reward": 0.7851440906524658, "reward_std": 0.03541453182697296, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10746843367815018, "sampling/sampling_logp_difference/max": 2.4928441047668457, "sampling/importance_sampling_ratio/min": 0.08267449587583542, "sampling/importance_sampling_ratio/mean": 1.0148011445999146, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5941140130162239, "clip_ratio/low_mean": 0.058562696911394596, "clip_ratio/low_min": 0.058562696911394596, "clip_ratio/high_mean": 0.05581792537122965, "clip_ratio/high_max": 0.05581792537122965, "clip_ratio/region_mean": 0.11438062228262424, "reward_total_mean": 0.7851440906524658, "reward_meter_mean": 0.9124640822410583, "reward_meter_std": 0.07928428798913956, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9853527545928955, "reward_repeat_soft_std": 0.01585215888917446, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7851440906524658, "reward_total_composite_std": 0.03541453182697296} {"timestamp_utc": "2026-04-13T03:57:06Z", "mode": "train", "global_step": 2245, "epoch": 0.22551481667503767, "loss": -0.1516, "grad_norm": 1.4892480373382568, "learning_rate": 3.2000000000000003e-06, "num_tokens": 4128387.0, "completions/mean_length": 118.75, "completions/min_length": 54.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 62.57143020629883, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.8746751546859741, "rewards/meter/std": 0.3252800405025482, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9022780656814575, "rewards/repeat_soft/std": 0.21489737927913666, "rewards/judge_quality/mean": 0.39375001192092896, "rewards/judge_quality/std": 0.25048166513442993, "rewards/total_composite/mean": 0.7153788805007935, "rewards/total_composite/std": 0.2965329885482788, "reward": 0.7153788805007935, "reward_std": 0.2965329587459564, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11534709483385086, "sampling/sampling_logp_difference/max": 1.441070556640625, "sampling/importance_sampling_ratio/min": 0.2366742491722107, "sampling/importance_sampling_ratio/mean": 1.0122485160827637, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8127814158797264, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09421163611114025, "clip_ratio/high_max": 0.09421163611114025, "clip_ratio/region_mean": 0.09421163611114025, "reward_total_mean": 0.7153788805007935, "reward_meter_mean": 0.8746751546859741, "reward_meter_std": 0.3252800405025482, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9022780656814575, "reward_repeat_soft_std": 0.21489737927913666, "reward_judge_quality_mean": 0.39375001192092896, "reward_judge_quality_std": 0.25048166513442993, "reward_total_composite_mean": 0.7153788805007935, "reward_total_composite_std": 0.2965329885482788} {"timestamp_utc": "2026-04-13T03:57:17Z", "mode": "train", "global_step": 2246, "epoch": 0.22561526870919135, "loss": -0.1913, "grad_norm": 2.319641351699829, "learning_rate": 3.196969696969697e-06, "num_tokens": 4130465.0, "completions/mean_length": 213.75, "completions/min_length": 98.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 114.33333587646484, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.6775188446044922, "rewards/meter/std": 0.4139173924922943, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.36443448066711426, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.994793176651001, "rewards/repeat_soft/std": 0.005672558210790157, "rewards/judge_quality/mean": 0.35874998569488525, "rewards/judge_quality/std": 0.23092593252658844, "rewards/total_composite/mean": 0.5410014390945435, "rewards/total_composite/std": 0.36445334553718567, "reward": 0.5410014390945435, "reward_std": 0.3644533157348633, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14365896582603455, "sampling/sampling_logp_difference/max": 2.1669881343841553, "sampling/importance_sampling_ratio/min": 0.11452202498912811, "sampling/importance_sampling_ratio/mean": 1.0100346803665161, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6026045382022858, "clip_ratio/low_mean": 0.015046296641230583, "clip_ratio/low_min": 0.015046296641230583, "clip_ratio/high_mean": 0.08097221795469522, "clip_ratio/high_max": 0.08097221795469522, "clip_ratio/region_mean": 0.09601851459592581, "reward_total_mean": 0.5410014390945435, "reward_meter_mean": 0.6775188446044922, "reward_meter_std": 0.4139173924922943, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.36443448066711426, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.994793176651001, "reward_repeat_soft_std": 0.005672558210790157, "reward_judge_quality_mean": 0.35874998569488525, "reward_judge_quality_std": 0.23092593252658844, "reward_total_composite_mean": 0.5410014390945435, "reward_total_composite_std": 0.36445334553718567} {"timestamp_utc": "2026-04-13T03:57:29Z", "mode": "train", "global_step": 2247, "epoch": 0.22571572074334506, "loss": -0.1556, "grad_norm": 1.5124938488006592, "learning_rate": 3.1939393939393944e-06, "num_tokens": 4132107.0, "completions/mean_length": 116.25, "completions/min_length": 53.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 59.71428680419922, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9545638561248779, "rewards/meter/std": 0.03747313469648361, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.94983971118927, "rewards/repeat_soft/std": 0.05104057863354683, "rewards/judge_quality/mean": 0.8112499713897705, "rewards/judge_quality/std": 0.3075914680957794, "rewards/total_composite/mean": 0.8340286612510681, "rewards/total_composite/std": 0.3374604284763336, "reward": 0.8340286612510681, "reward_std": 0.3374604284763336, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10942684859037399, "sampling/sampling_logp_difference/max": 3.9109694957733154, "sampling/importance_sampling_ratio/min": 0.02002108097076416, "sampling/importance_sampling_ratio/mean": 1.0019495487213135, "sampling/importance_sampling_ratio/max": 1.8379110097885132, "entropy": 0.48923107981681824, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10292004980146885, "clip_ratio/high_max": 0.10292004980146885, "clip_ratio/region_mean": 0.10292004980146885, "reward_total_mean": 0.8340286612510681, "reward_meter_mean": 0.9545638561248779, "reward_meter_std": 0.03747313469648361, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.94983971118927, "reward_repeat_soft_std": 0.05104057863354683, "reward_judge_quality_mean": 0.8112499713897705, "reward_judge_quality_std": 0.3075914680957794, "reward_total_composite_mean": 0.8340286612510681, "reward_total_composite_std": 0.3374604284763336} {"timestamp_utc": "2026-04-13T03:57:35Z", "mode": "train", "global_step": 2248, "epoch": 0.22581617277749874, "loss": 0.0296, "grad_norm": 10.904314041137695, "learning_rate": 3.190909090909091e-06, "num_tokens": 4133860.0, "completions/mean_length": 63.125, "completions/min_length": 51.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.125, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9079767465591431, "rewards/meter/std": 0.24030274152755737, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9721095561981201, "rewards/repeat_soft/std": 0.02584938332438469, "rewards/judge_quality/mean": 0.4725000262260437, "rewards/judge_quality/std": 0.1011011004447937, "rewards/total_composite/mean": 0.7975504398345947, "rewards/total_composite/std": 0.07871268689632416, "reward": 0.7975504398345947, "reward_std": 0.07871268689632416, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11001181602478027, "sampling/sampling_logp_difference/max": 1.232893466949463, "sampling/importance_sampling_ratio/min": 0.29144805669784546, "sampling/importance_sampling_ratio/mean": 1.0225958824157715, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.816646046936512, "clip_ratio/low_mean": 0.007575757801532745, "clip_ratio/low_min": 0.007575757801532745, "clip_ratio/high_mean": 0.08995268819853663, "clip_ratio/high_max": 0.08995268819853663, "clip_ratio/region_mean": 0.09752844600006938, "reward_total_mean": 0.7975504398345947, "reward_meter_mean": 0.9079767465591431, "reward_meter_std": 0.24030274152755737, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9721095561981201, "reward_repeat_soft_std": 0.02584938332438469, "reward_judge_quality_mean": 0.4725000262260437, "reward_judge_quality_std": 0.1011011004447937, "reward_total_composite_mean": 0.7975504398345947, "reward_total_composite_std": 0.07871268689632416} {"timestamp_utc": "2026-04-13T03:57:42Z", "mode": "train", "global_step": 2249, "epoch": 0.22591662481165242, "loss": 0.0212, "grad_norm": 7.783763885498047, "learning_rate": 3.187878787878788e-06, "num_tokens": 4135932.0, "completions/mean_length": 106.0, "completions/min_length": 91.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.0, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.7384071350097656, "rewards/meter/std": 0.2626400589942932, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624267220497131, "rewards/repeat_soft/std": 0.0155600905418396, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.7270258665084839, "rewards/total_composite/std": 0.118662990629673, "reward": 0.7270258665084839, "reward_std": 0.1186629980802536, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12397383898496628, "sampling/sampling_logp_difference/max": 1.5126490592956543, "sampling/importance_sampling_ratio/min": 0.2203255593776703, "sampling/importance_sampling_ratio/mean": 1.0113738775253296, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7540765814483166, "clip_ratio/low_mean": 0.05226615630090237, "clip_ratio/low_min": 0.05226615630090237, "clip_ratio/high_mean": 0.059753126464784145, "clip_ratio/high_max": 0.059753126464784145, "clip_ratio/region_mean": 0.11201928276568651, "reward_total_mean": 0.7270258665084839, "reward_meter_mean": 0.7384071350097656, "reward_meter_std": 0.2626400589942932, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624267220497131, "reward_repeat_soft_std": 0.0155600905418396, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.7270258665084839, "reward_total_composite_std": 0.118662990629673} {"timestamp_utc": "2026-04-13T03:57:49Z", "mode": "train", "global_step": 2250, "epoch": 0.22601707684580613, "loss": -0.0239, "grad_norm": 8.644647598266602, "learning_rate": 3.1848484848484853e-06, "num_tokens": 4137614.0, "completions/mean_length": 54.25, "completions/min_length": 51.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.25, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9853389263153076, "rewards/meter/std": 0.010937406681478024, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8949851393699646, "rewards/repeat_soft/std": 0.09586358070373535, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.8299009799957275, "rewards/total_composite/std": 0.05599089711904526, "reward": 0.8299009799957275, "reward_std": 0.05599091947078705, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11056207120418549, "sampling/sampling_logp_difference/max": 1.5479373931884766, "sampling/importance_sampling_ratio/min": 0.2126862108707428, "sampling/importance_sampling_ratio/mean": 1.006754755973816, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5486471876502037, "clip_ratio/low_mean": 0.06486641871742904, "clip_ratio/low_min": 0.06486641871742904, "clip_ratio/high_mean": 0.0062500000931322575, "clip_ratio/high_max": 0.0062500000931322575, "clip_ratio/region_mean": 0.0711164188105613, "reward_total_mean": 0.8299009799957275, "reward_meter_mean": 0.9853389263153076, "reward_meter_std": 0.010937406681478024, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8949851393699646, "reward_repeat_soft_std": 0.09586358070373535, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.8299009799957275, "reward_total_composite_std": 0.05599089711904526} {"timestamp_utc": "2026-04-13T03:58:43Z", "mode": "eval", "global_step": 2250, "epoch": 0.22601707684580613, "eval_loss": NaN, "eval_runtime": 54.2965, "eval_samples_per_second": 1.473, "eval_steps_per_second": 0.184, "eval_num_tokens": 4137614.0, "eval_completions/mean_length": 108.9375, "eval_completions/min_length": 43.2, "eval_completions/max_length": 232.0, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 98.85, "eval_completions/min_terminated_length": 43.2, "eval_completions/max_terminated_length": 201.2, "eval_rewards/meter/mean": 0.8452246189117432, "eval_rewards/meter/std": 0.1936796586960554, "eval_rewards/count_adherence/mean": 0.964166671037674, "eval_rewards/count_adherence/std": 0.0773802638053894, "eval_rewards/hard_gate/mean": 0.95, "eval_rewards/hard_gate/std": 0.08711026012897491, "eval_rewards/repeat_soft/mean": 0.9561155259609222, "eval_rewards/repeat_soft/std": 0.03797896355390549, "eval_rewards/judge_quality/mean": 0.4248749941587448, "eval_rewards/judge_quality/std": 0.16630513966083527, "eval_rewards/total_composite/mean": 0.7225155472755432, "eval_rewards/total_composite/std": 0.14284334890544415, "eval_reward": 0.7225155472755432, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.06406422965228557, "eval_sampling/sampling_logp_difference/max": 1.0373783588409424, "eval_sampling/importance_sampling_ratio/min": 0.36141542345285416, "eval_sampling/importance_sampling_ratio/mean": 1.0138389587402343, "eval_sampling/importance_sampling_ratio/max": 1.4221213817596436, "eval_entropy": 1.1553454637527465, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7225155472755432, "eval_reward_meter_mean": 0.8452246189117432, "eval_reward_meter_std": 0.1936796586960554, "eval_reward_count_adherence_mean": 0.964166671037674, "eval_reward_count_adherence_std": 0.0773802638053894, "eval_reward_hard_gate_mean": 0.95, "eval_reward_hard_gate_std": 0.08711026012897491, "eval_reward_repeat_soft_mean": 0.9561155259609222, "eval_reward_repeat_soft_std": 0.03797896355390549, "eval_reward_judge_quality_mean": 0.4248749941587448, "eval_reward_judge_quality_std": 0.16630513966083527, "eval_reward_total_composite_mean": 0.7225155472755432, "eval_reward_total_composite_std": 0.14284334890544415} {"timestamp_utc": "2026-04-13T03:58:54Z", "mode": "train", "global_step": 2251, "epoch": 0.2261175288799598, "loss": 0.008, "grad_norm": 8.82497501373291, "learning_rate": 3.181818181818182e-06, "num_tokens": 4139616.0, "completions/mean_length": 79.25, "completions/min_length": 73.0, "completions/max_length": 89.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.25, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 89.0, "rewards/meter/mean": 0.9832751750946045, "rewards/meter/std": 0.011813981458544731, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9221707582473755, "rewards/repeat_soft/std": 0.059747472405433655, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7093109488487244, "rewards/total_composite/std": 0.28665101528167725, "reward": 0.7093109488487244, "reward_std": 0.28665101528167725, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10817835479974747, "sampling/sampling_logp_difference/max": 1.5828371047973633, "sampling/importance_sampling_ratio/min": 0.2053915411233902, "sampling/importance_sampling_ratio/mean": 1.0154736042022705, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5309414938092232, "clip_ratio/low_mean": 0.017628205940127373, "clip_ratio/low_min": 0.017628205940127373, "clip_ratio/high_mean": 0.09351559448987246, "clip_ratio/high_max": 0.09351559448987246, "clip_ratio/region_mean": 0.11114380042999983, "reward_total_mean": 0.7093109488487244, "reward_meter_mean": 0.9832751750946045, "reward_meter_std": 0.011813981458544731, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9221707582473755, "reward_repeat_soft_std": 0.059747472405433655, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7093109488487244, "reward_total_composite_std": 0.28665101528167725} {"timestamp_utc": "2026-04-13T03:59:02Z", "mode": "train", "global_step": 2252, "epoch": 0.22621798091411352, "loss": 0.0343, "grad_norm": 11.428426742553711, "learning_rate": 3.178787878787879e-06, "num_tokens": 4141415.0, "completions/mean_length": 61.875, "completions/min_length": 56.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.875, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.657579779624939, "rewards/meter/std": 0.37043097615242004, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.978988766670227, "rewards/repeat_soft/std": 0.010480940341949463, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.6908098459243774, "rewards/total_composite/std": 0.18616268038749695, "reward": 0.6908098459243774, "reward_std": 0.18616269528865814, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11076050251722336, "sampling/sampling_logp_difference/max": 1.2837247848510742, "sampling/importance_sampling_ratio/min": 0.27700358629226685, "sampling/importance_sampling_ratio/mean": 1.0067191123962402, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6810010522603989, "clip_ratio/low_mean": 0.024653309723362327, "clip_ratio/low_min": 0.024653309723362327, "clip_ratio/high_mean": 0.06443139910697937, "clip_ratio/high_max": 0.06443139910697937, "clip_ratio/region_mean": 0.0890847088303417, "reward_total_mean": 0.6908098459243774, "reward_meter_mean": 0.657579779624939, "reward_meter_std": 0.37043097615242004, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.978988766670227, "reward_repeat_soft_std": 0.010480940341949463, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.6908098459243774, "reward_total_composite_std": 0.18616268038749695} {"timestamp_utc": "2026-04-13T03:59:13Z", "mode": "train", "global_step": 2253, "epoch": 0.2263184329482672, "loss": -0.1477, "grad_norm": 1.7602148056030273, "learning_rate": 3.1757575757575758e-06, "num_tokens": 4143251.0, "completions/mean_length": 112.5, "completions/min_length": 48.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 55.42857360839844, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.7953401803970337, "rewards/meter/std": 0.3364420235157013, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9649068713188171, "rewards/repeat_soft/std": 0.04311063513159752, "rewards/judge_quality/mean": 0.39249998331069946, "rewards/judge_quality/std": 0.13905291259288788, "rewards/total_composite/mean": 0.6890187859535217, "rewards/total_composite/std": 0.281872421503067, "reward": 0.6890187859535217, "reward_std": 0.281872421503067, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12188266962766647, "sampling/sampling_logp_difference/max": 1.5118813514709473, "sampling/importance_sampling_ratio/min": 0.22049476206302643, "sampling/importance_sampling_ratio/mean": 0.985183596611023, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6630796752870083, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.12687543500214815, "clip_ratio/high_max": 0.12687543500214815, "clip_ratio/region_mean": 0.12687543500214815, "reward_total_mean": 0.6890187859535217, "reward_meter_mean": 0.7953401803970337, "reward_meter_std": 0.3364420235157013, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9649068713188171, "reward_repeat_soft_std": 0.04311063513159752, "reward_judge_quality_mean": 0.39249998331069946, "reward_judge_quality_std": 0.13905291259288788, "reward_total_composite_mean": 0.6890187859535217, "reward_total_composite_std": 0.281872421503067} {"timestamp_utc": "2026-04-13T03:59:20Z", "mode": "train", "global_step": 2254, "epoch": 0.22641888498242088, "loss": 0.0161, "grad_norm": 15.954621315002441, "learning_rate": 3.172727272727273e-06, "num_tokens": 4144696.0, "completions/mean_length": 27.625, "completions/min_length": 25.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.625, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.9861001372337341, "rewards/meter/std": 0.005697230342775583, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9508087635040283, "rewards/repeat_soft/std": 0.019656365737318993, "rewards/judge_quality/mean": 0.4300000071525574, "rewards/judge_quality/std": 0.21993505954742432, "rewards/total_composite/mean": 0.8178259134292603, "rewards/total_composite/std": 0.06779871135950089, "reward": 0.8178259134292603, "reward_std": 0.06779871881008148, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11544335633516312, "sampling/sampling_logp_difference/max": 1.406807780265808, "sampling/importance_sampling_ratio/min": 0.2449238896369934, "sampling/importance_sampling_ratio/mean": 1.0286731719970703, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6729878261685371, "clip_ratio/low_mean": 0.03682123916223645, "clip_ratio/low_min": 0.03682123916223645, "clip_ratio/high_mean": 0.0439495537430048, "clip_ratio/high_max": 0.0439495537430048, "clip_ratio/region_mean": 0.08077079290524125, "reward_total_mean": 0.8178259134292603, "reward_meter_mean": 0.9861001372337341, "reward_meter_std": 0.005697230342775583, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9508087635040283, "reward_repeat_soft_std": 0.019656365737318993, "reward_judge_quality_mean": 0.4300000071525574, "reward_judge_quality_std": 0.21993505954742432, "reward_total_composite_mean": 0.8178259134292603, "reward_total_composite_std": 0.06779871135950089} {"timestamp_utc": "2026-04-13T03:59:26Z", "mode": "train", "global_step": 2255, "epoch": 0.2265193370165746, "loss": 0.0754, "grad_norm": 16.069299697875977, "learning_rate": 3.16969696969697e-06, "num_tokens": 4146120.0, "completions/mean_length": 28.0, "completions/min_length": 23.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.0, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9062232971191406, "rewards/meter/std": 0.21746006608009338, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9497413039207458, "rewards/repeat_soft/std": 0.023724183440208435, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.7768995761871338, "rewards/total_composite/std": 0.1007051169872284, "reward": 0.7768995761871338, "reward_std": 0.1007051169872284, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10776123404502869, "sampling/sampling_logp_difference/max": 1.3843307495117188, "sampling/importance_sampling_ratio/min": 0.2504914104938507, "sampling/importance_sampling_ratio/mean": 1.0206304788589478, "sampling/importance_sampling_ratio/max": 1.995580792427063, "entropy": 0.8015879690647125, "clip_ratio/low_mean": 0.027812499552965164, "clip_ratio/low_min": 0.027812499552965164, "clip_ratio/high_mean": 0.07771096518263221, "clip_ratio/high_max": 0.07771096518263221, "clip_ratio/region_mean": 0.10552346473559737, "reward_total_mean": 0.7768995761871338, "reward_meter_mean": 0.9062232971191406, "reward_meter_std": 0.21746006608009338, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9497413039207458, "reward_repeat_soft_std": 0.023724183440208435, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.7768995761871338, "reward_total_composite_std": 0.1007051169872284} {"timestamp_utc": "2026-04-13T03:59:33Z", "mode": "train", "global_step": 2256, "epoch": 0.22661978905072827, "loss": 0.0142, "grad_norm": 10.7279691696167, "learning_rate": 3.1666666666666667e-06, "num_tokens": 4147910.0, "completions/mean_length": 52.75, "completions/min_length": 44.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.75, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9731703996658325, "rewards/meter/std": 0.03464793041348457, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9906036853790283, "rewards/repeat_soft/std": 0.011412560939788818, "rewards/judge_quality/mean": 0.5149999856948853, "rewards/judge_quality/std": 0.16017849743366241, "rewards/total_composite/mean": 0.8414870500564575, "rewards/total_composite/std": 0.05398501083254814, "reward": 0.8414870500564575, "reward_std": 0.05398501455783844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12129145115613937, "sampling/sampling_logp_difference/max": 1.8709778785705566, "sampling/importance_sampling_ratio/min": 0.15397301316261292, "sampling/importance_sampling_ratio/mean": 1.01751708984375, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8325718194246292, "clip_ratio/low_mean": 0.08698830101639032, "clip_ratio/low_min": 0.08698830101639032, "clip_ratio/high_mean": 0.03489583358168602, "clip_ratio/high_max": 0.03489583358168602, "clip_ratio/region_mean": 0.12188413459807634, "reward_total_mean": 0.8414870500564575, "reward_meter_mean": 0.9731703996658325, "reward_meter_std": 0.03464793041348457, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9906036853790283, "reward_repeat_soft_std": 0.011412560939788818, "reward_judge_quality_mean": 0.5149999856948853, "reward_judge_quality_std": 0.16017849743366241, "reward_total_composite_mean": 0.8414870500564575, "reward_total_composite_std": 0.05398501083254814} {"timestamp_utc": "2026-04-13T03:59:40Z", "mode": "train", "global_step": 2257, "epoch": 0.22672024108488198, "loss": 0.0368, "grad_norm": 7.6819891929626465, "learning_rate": 3.1636363636363635e-06, "num_tokens": 4150176.0, "completions/mean_length": 113.25, "completions/min_length": 104.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.25, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.8782305121421814, "rewards/meter/std": 0.231111079454422, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8726551532745361, "rewards/repeat_soft/std": 0.06656970828771591, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7457192540168762, "rewards/total_composite/std": 0.12364662438631058, "reward": 0.7457192540168762, "reward_std": 0.12364663183689117, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10107118636369705, "sampling/sampling_logp_difference/max": 1.5555038452148438, "sampling/importance_sampling_ratio/min": 0.21108299493789673, "sampling/importance_sampling_ratio/mean": 1.0066232681274414, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5096354782581329, "clip_ratio/low_mean": 0.016361556015908718, "clip_ratio/low_min": 0.016361556015908718, "clip_ratio/high_mean": 0.08660990744829178, "clip_ratio/high_max": 0.08660990744829178, "clip_ratio/region_mean": 0.1029714634642005, "reward_total_mean": 0.7457192540168762, "reward_meter_mean": 0.8782305121421814, "reward_meter_std": 0.231111079454422, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8726551532745361, "reward_repeat_soft_std": 0.06656970828771591, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7457192540168762, "reward_total_composite_std": 0.12364662438631058} {"timestamp_utc": "2026-04-13T03:59:47Z", "mode": "train", "global_step": 2258, "epoch": 0.22682069311903566, "loss": -0.0196, "grad_norm": 10.037454605102539, "learning_rate": 3.1606060606060608e-06, "num_tokens": 4151882.0, "completions/mean_length": 58.25, "completions/min_length": 48.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.25, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9737803936004639, "rewards/meter/std": 0.04565475881099701, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9381158351898193, "rewards/repeat_soft/std": 0.055768005549907684, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.8642628192901611, "rewards/total_composite/std": 0.06823773682117462, "reward": 0.8642628192901611, "reward_std": 0.06823773682117462, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10508778691291809, "sampling/sampling_logp_difference/max": 2.353466033935547, "sampling/importance_sampling_ratio/min": 0.09503918141126633, "sampling/importance_sampling_ratio/mean": 1.0062452554702759, "sampling/importance_sampling_ratio/max": 1.9735544919967651, "entropy": 0.6156709901988506, "clip_ratio/low_mean": 0.05278379190713167, "clip_ratio/low_min": 0.05278379190713167, "clip_ratio/high_mean": 0.039584191516041756, "clip_ratio/high_max": 0.039584191516041756, "clip_ratio/region_mean": 0.09236798342317343, "reward_total_mean": 0.8642628192901611, "reward_meter_mean": 0.9737803936004639, "reward_meter_std": 0.04565475881099701, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9381158351898193, "reward_repeat_soft_std": 0.055768005549907684, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.8642628192901611, "reward_total_composite_std": 0.06823773682117462} {"timestamp_utc": "2026-04-13T03:59:58Z", "mode": "train", "global_step": 2259, "epoch": 0.22692114515318934, "loss": -0.1426, "grad_norm": 2.5789902210235596, "learning_rate": 3.1575757575757576e-06, "num_tokens": 4153620.0, "completions/mean_length": 119.25, "completions/min_length": 57.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 63.142860412597656, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.6876217722892761, "rewards/meter/std": 0.3667049705982208, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9709495306015015, "rewards/repeat_soft/std": 0.018462274223566055, "rewards/judge_quality/mean": 0.6237499713897705, "rewards/judge_quality/std": 0.33907175064086914, "rewards/total_composite/mean": 0.7022069692611694, "rewards/total_composite/std": 0.3041611313819885, "reward": 0.7022069692611694, "reward_std": 0.30416110157966614, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09542202949523926, "sampling/sampling_logp_difference/max": 1.405961036682129, "sampling/importance_sampling_ratio/min": 0.24513137340545654, "sampling/importance_sampling_ratio/mean": 0.9939407706260681, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4825702831149101, "clip_ratio/low_mean": 0.003968254197388887, "clip_ratio/low_min": 0.003968254197388887, "clip_ratio/high_mean": 0.07778538949787617, "clip_ratio/high_max": 0.07778538949787617, "clip_ratio/region_mean": 0.08175364369526505, "reward_total_mean": 0.7022069692611694, "reward_meter_mean": 0.6876217722892761, "reward_meter_std": 0.3667049705982208, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9709495306015015, "reward_repeat_soft_std": 0.018462274223566055, "reward_judge_quality_mean": 0.6237499713897705, "reward_judge_quality_std": 0.33907175064086914, "reward_total_composite_mean": 0.7022069692611694, "reward_total_composite_std": 0.3041611313819885} {"timestamp_utc": "2026-04-13T04:00:05Z", "mode": "train", "global_step": 2260, "epoch": 0.22702159718734305, "loss": 0.0416, "grad_norm": 20.960084915161133, "learning_rate": 3.1545454545454545e-06, "num_tokens": 4155235.0, "completions/mean_length": 46.875, "completions/min_length": 43.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.875, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.9903333187103271, "rewards/meter/std": 0.0031657028011977673, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9576224088668823, "rewards/repeat_soft/std": 0.032557349652051926, "rewards/judge_quality/mean": 0.44624999165534973, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8252872228622437, "rewards/total_composite/std": 0.003904202952980995, "reward": 0.8252872228622437, "reward_std": 0.0039041992276906967, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0861687883734703, "sampling/sampling_logp_difference/max": 1.4165947437286377, "sampling/importance_sampling_ratio/min": 0.33228006958961487, "sampling/importance_sampling_ratio/mean": 1.0128660202026367, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45458413287997246, "clip_ratio/low_mean": 0.04241653555072844, "clip_ratio/low_min": 0.04241653555072844, "clip_ratio/high_mean": 0.0193482325412333, "clip_ratio/high_max": 0.0193482325412333, "clip_ratio/region_mean": 0.06176476809196174, "reward_total_mean": 0.8252872228622437, "reward_meter_mean": 0.9903333187103271, "reward_meter_std": 0.0031657028011977673, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9576224088668823, "reward_repeat_soft_std": 0.032557349652051926, "reward_judge_quality_mean": 0.44624999165534973, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8252872228622437, "reward_total_composite_std": 0.003904202952980995} {"timestamp_utc": "2026-04-13T04:00:13Z", "mode": "train", "global_step": 2261, "epoch": 0.22712204922149673, "loss": 0.0497, "grad_norm": 6.575899600982666, "learning_rate": 3.1515151515151517e-06, "num_tokens": 4158156.0, "completions/mean_length": 159.125, "completions/min_length": 148.0, "completions/max_length": 173.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 159.125, "completions/min_terminated_length": 148.0, "completions/max_terminated_length": 173.0, "rewards/meter/mean": 0.9558120965957642, "rewards/meter/std": 0.03580067679286003, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.918854296207428, "rewards/repeat_soft/std": 0.02509351260960102, "rewards/judge_quality/mean": 0.23374998569488525, "rewards/judge_quality/std": 0.09006941318511963, "rewards/total_composite/mean": 0.7421258687973022, "rewards/total_composite/std": 0.040580280125141144, "reward": 0.7421258687973022, "reward_std": 0.040580276399850845, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1179824247956276, "sampling/sampling_logp_difference/max": 2.9100985527038574, "sampling/importance_sampling_ratio/min": 0.05447036400437355, "sampling/importance_sampling_ratio/mean": 0.9949078559875488, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5630434155464172, "clip_ratio/low_mean": 0.03514751046895981, "clip_ratio/low_min": 0.03514751046895981, "clip_ratio/high_mean": 0.07116651069372892, "clip_ratio/high_max": 0.07116651069372892, "clip_ratio/region_mean": 0.10631402116268873, "reward_total_mean": 0.7421258687973022, "reward_meter_mean": 0.9558120965957642, "reward_meter_std": 0.03580067679286003, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.918854296207428, "reward_repeat_soft_std": 0.02509351260960102, "reward_judge_quality_mean": 0.23374998569488525, "reward_judge_quality_std": 0.09006941318511963, "reward_total_composite_mean": 0.7421258687973022, "reward_total_composite_std": 0.040580280125141144} {"timestamp_utc": "2026-04-13T04:00:24Z", "mode": "train", "global_step": 2262, "epoch": 0.22722250125565044, "loss": -0.1681, "grad_norm": 2.2337734699249268, "learning_rate": 3.1484848484848485e-06, "num_tokens": 4160053.0, "completions/mean_length": 134.125, "completions/min_length": 75.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 80.14286041259766, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.8564227819442749, "rewards/meter/std": 0.3461928069591522, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9036253690719604, "rewards/repeat_soft/std": 0.08442997932434082, "rewards/judge_quality/mean": 0.4475000202655792, "rewards/judge_quality/std": 0.32371723651885986, "rewards/total_composite/mean": 0.7268778085708618, "rewards/total_composite/std": 0.30780646204948425, "reward": 0.7268778085708618, "reward_std": 0.30780646204948425, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10470791906118393, "sampling/sampling_logp_difference/max": 1.411586046218872, "sampling/importance_sampling_ratio/min": 0.24375636875629425, "sampling/importance_sampling_ratio/mean": 0.9934898614883423, "sampling/importance_sampling_ratio/max": 1.7231965065002441, "entropy": 0.5335138849914074, "clip_ratio/low_mean": 0.008620689623057842, "clip_ratio/low_min": 0.008620689623057842, "clip_ratio/high_mean": 0.09349443484097719, "clip_ratio/high_max": 0.09349443484097719, "clip_ratio/region_mean": 0.10211512446403503, "reward_total_mean": 0.7268778085708618, "reward_meter_mean": 0.8564227819442749, "reward_meter_std": 0.3461928069591522, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9036253690719604, "reward_repeat_soft_std": 0.08442997932434082, "reward_judge_quality_mean": 0.4475000202655792, "reward_judge_quality_std": 0.32371723651885986, "reward_total_composite_mean": 0.7268778085708618, "reward_total_composite_std": 0.30780646204948425} {"timestamp_utc": "2026-04-13T04:00:38Z", "mode": "train", "global_step": 2263, "epoch": 0.22732295328980412, "loss": -0.1011, "grad_norm": 3.564178705215454, "learning_rate": 3.145454545454546e-06, "num_tokens": 4161873.0, "completions/mean_length": 123.5, "completions/min_length": 61.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.7270520925521851, "rewards/meter/std": 0.40972214937210083, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9094168543815613, "rewards/repeat_soft/std": 0.049157802015542984, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.6598026156425476, "rewards/total_composite/std": 0.2385053187608719, "reward": 0.6598026156425476, "reward_std": 0.2385053187608719, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13131815195083618, "sampling/sampling_logp_difference/max": 2.230424404144287, "sampling/importance_sampling_ratio/min": 0.10748279839754105, "sampling/importance_sampling_ratio/mean": 0.9972248673439026, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36464980989694595, "clip_ratio/low_mean": 0.0193452388048172, "clip_ratio/low_min": 0.0193452388048172, "clip_ratio/high_mean": 0.07764608366414905, "clip_ratio/high_max": 0.07764608366414905, "clip_ratio/region_mean": 0.09699132246896625, "reward_total_mean": 0.6598026156425476, "reward_meter_mean": 0.7270520925521851, "reward_meter_std": 0.40972214937210083, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.2651650309562683, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9094168543815613, "reward_repeat_soft_std": 0.049157802015542984, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.6598026156425476, "reward_total_composite_std": 0.2385053187608719} {"timestamp_utc": "2026-04-13T04:00:44Z", "mode": "train", "global_step": 2264, "epoch": 0.2274234053239578, "loss": 0.0785, "grad_norm": 9.386005401611328, "learning_rate": 3.142424242424243e-06, "num_tokens": 4163890.0, "completions/mean_length": 60.125, "completions/min_length": 52.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.125, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9586050510406494, "rewards/meter/std": 0.07523561269044876, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9734928011894226, "rewards/repeat_soft/std": 0.027708159759640694, "rewards/judge_quality/mean": 0.5862500071525574, "rewards/judge_quality/std": 0.2822834253311157, "rewards/total_composite/mean": 0.8545965552330017, "rewards/total_composite/std": 0.1002713143825531, "reward": 0.8545965552330017, "reward_std": 0.1002713143825531, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11590289324522018, "sampling/sampling_logp_difference/max": 1.250197410583496, "sampling/importance_sampling_ratio/min": 0.28644824028015137, "sampling/importance_sampling_ratio/mean": 0.995917558670044, "sampling/importance_sampling_ratio/max": 1.9777640104293823, "entropy": 0.6354683600366116, "clip_ratio/low_mean": 0.06375715998001397, "clip_ratio/low_min": 0.06375715998001397, "clip_ratio/high_mean": 0.02933524874970317, "clip_ratio/high_max": 0.02933524874970317, "clip_ratio/region_mean": 0.09309240872971714, "reward_total_mean": 0.8545965552330017, "reward_meter_mean": 0.9586050510406494, "reward_meter_std": 0.07523561269044876, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9734928011894226, "reward_repeat_soft_std": 0.027708159759640694, "reward_judge_quality_mean": 0.5862500071525574, "reward_judge_quality_std": 0.2822834253311157, "reward_total_composite_mean": 0.8545965552330017, "reward_total_composite_std": 0.1002713143825531} {"timestamp_utc": "2026-04-13T04:00:51Z", "mode": "train", "global_step": 2265, "epoch": 0.2275238573581115, "loss": -0.0316, "grad_norm": 6.340027332305908, "learning_rate": 3.13939393939394e-06, "num_tokens": 4165606.0, "completions/mean_length": 56.5, "completions/min_length": 50.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.5, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9095158576965332, "rewards/meter/std": 0.22569316625595093, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9829150438308716, "rewards/repeat_soft/std": 0.014243262819945812, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.7880736589431763, "rewards/total_composite/std": 0.09928301721811295, "reward": 0.7880736589431763, "reward_std": 0.09928300976753235, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09756655246019363, "sampling/sampling_logp_difference/max": 1.9538869857788086, "sampling/importance_sampling_ratio/min": 0.14172212779521942, "sampling/importance_sampling_ratio/mean": 1.0006507635116577, "sampling/importance_sampling_ratio/max": 1.745624303817749, "entropy": 0.5603644028306007, "clip_ratio/low_mean": 0.012500000186264515, "clip_ratio/low_min": 0.012500000186264515, "clip_ratio/high_mean": 0.08324253791943192, "clip_ratio/high_max": 0.08324253791943192, "clip_ratio/region_mean": 0.09574253810569644, "reward_total_mean": 0.7880736589431763, "reward_meter_mean": 0.9095158576965332, "reward_meter_std": 0.22569316625595093, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9829150438308716, "reward_repeat_soft_std": 0.014243262819945812, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.7880736589431763, "reward_total_composite_std": 0.09928301721811295} {"timestamp_utc": "2026-04-13T04:00:58Z", "mode": "train", "global_step": 2266, "epoch": 0.2276243093922652, "loss": 0.0004, "grad_norm": 10.427947044372559, "learning_rate": 3.1363636363636367e-06, "num_tokens": 4167314.0, "completions/mean_length": 59.5, "completions/min_length": 52.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.5, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.8726713061332703, "rewards/meter/std": 0.3079448938369751, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9879286289215088, "rewards/repeat_soft/std": 0.015222122892737389, "rewards/judge_quality/mean": 0.6599999666213989, "rewards/judge_quality/std": 0.2392846643924713, "rewards/total_composite/mean": 0.8394949436187744, "rewards/total_composite/std": 0.17600899934768677, "reward": 0.8394949436187744, "reward_std": 0.17600898444652557, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11503618210554123, "sampling/sampling_logp_difference/max": 1.9907522201538086, "sampling/importance_sampling_ratio/min": 0.13659264147281647, "sampling/importance_sampling_ratio/mean": 0.9859039783477783, "sampling/importance_sampling_ratio/max": 1.6505321264266968, "entropy": 0.654788427054882, "clip_ratio/low_mean": 0.0645125713199377, "clip_ratio/low_min": 0.0645125713199377, "clip_ratio/high_mean": 0.05872596614062786, "clip_ratio/high_max": 0.05872596614062786, "clip_ratio/region_mean": 0.12323853746056557, "reward_total_mean": 0.8394949436187744, "reward_meter_mean": 0.8726713061332703, "reward_meter_std": 0.3079448938369751, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9879286289215088, "reward_repeat_soft_std": 0.015222122892737389, "reward_judge_quality_mean": 0.6599999666213989, "reward_judge_quality_std": 0.2392846643924713, "reward_total_composite_mean": 0.8394949436187744, "reward_total_composite_std": 0.17600899934768677} {"timestamp_utc": "2026-04-13T04:01:04Z", "mode": "train", "global_step": 2267, "epoch": 0.2277247614264189, "loss": 0.0266, "grad_norm": 9.67420768737793, "learning_rate": 3.133333333333334e-06, "num_tokens": 4168842.0, "completions/mean_length": 43.0, "completions/min_length": 41.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.0, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.9918324947357178, "rewards/meter/std": 0.0035113831982016563, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9407894015312195, "rewards/repeat_soft/std": 0.04554540291428566, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.10260014235973358, "rewards/total_composite/mean": 0.8310285806655884, "rewards/total_composite/std": 0.031583383679389954, "reward": 0.8310285806655884, "reward_std": 0.03158337622880936, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09011957794427872, "sampling/sampling_logp_difference/max": 1.2124838829040527, "sampling/importance_sampling_ratio/min": 0.2974575161933899, "sampling/importance_sampling_ratio/mean": 1.0077978372573853, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5041630119085312, "clip_ratio/low_mean": 0.08985098358243704, "clip_ratio/low_min": 0.08985098358243704, "clip_ratio/high_mean": 0.00872093066573143, "clip_ratio/high_max": 0.00872093066573143, "clip_ratio/region_mean": 0.09857191424816847, "reward_total_mean": 0.8310285806655884, "reward_meter_mean": 0.9918324947357178, "reward_meter_std": 0.0035113831982016563, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9407894015312195, "reward_repeat_soft_std": 0.04554540291428566, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.10260014235973358, "reward_total_composite_mean": 0.8310285806655884, "reward_total_composite_std": 0.031583383679389954} {"timestamp_utc": "2026-04-13T04:01:10Z", "mode": "train", "global_step": 2268, "epoch": 0.22782521346057258, "loss": 0.0327, "grad_norm": 13.918551445007324, "learning_rate": 3.130303030303031e-06, "num_tokens": 4170577.0, "completions/mean_length": 53.875, "completions/min_length": 49.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.875, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.8819011449813843, "rewards/meter/std": 0.2886991798877716, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9899546504020691, "rewards/repeat_soft/std": 0.012694615870714188, "rewards/judge_quality/mean": 0.5049999952316284, "rewards/judge_quality/std": 0.16801361739635468, "rewards/total_composite/mean": 0.7973510026931763, "rewards/total_composite/std": 0.14521536231040955, "reward": 0.7973510026931763, "reward_std": 0.14521536231040955, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11792677640914917, "sampling/sampling_logp_difference/max": 1.1506149768829346, "sampling/importance_sampling_ratio/min": 0.3164421021938324, "sampling/importance_sampling_ratio/mean": 1.0047249794006348, "sampling/importance_sampling_ratio/max": 1.9846330881118774, "entropy": 0.6404361538589001, "clip_ratio/low_mean": 0.013392857275903225, "clip_ratio/low_min": 0.013392857275903225, "clip_ratio/high_mean": 0.09150402899831533, "clip_ratio/high_max": 0.09150402899831533, "clip_ratio/region_mean": 0.10489688627421856, "reward_total_mean": 0.7973510026931763, "reward_meter_mean": 0.8819011449813843, "reward_meter_std": 0.2886991798877716, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9899546504020691, "reward_repeat_soft_std": 0.012694615870714188, "reward_judge_quality_mean": 0.5049999952316284, "reward_judge_quality_std": 0.16801361739635468, "reward_total_composite_mean": 0.7973510026931763, "reward_total_composite_std": 0.14521536231040955} {"timestamp_utc": "2026-04-13T04:01:21Z", "mode": "train", "global_step": 2269, "epoch": 0.22792566549472626, "loss": -0.1986, "grad_norm": 2.942647695541382, "learning_rate": 3.1272727272727276e-06, "num_tokens": 4173136.0, "completions/mean_length": 185.875, "completions/min_length": 117.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 139.2857208251953, "completions/min_terminated_length": 117.0, "completions/max_terminated_length": 154.0, "rewards/meter/mean": 0.4706726372241974, "rewards/meter/std": 0.3150758743286133, "rewards/count_adherence/mean": 0.8541666269302368, "rewards/count_adherence/std": 0.3500283658504486, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8487769961357117, "rewards/repeat_soft/std": 0.11049899458885193, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.1345893144607544, "rewards/total_composite/mean": 0.5034303665161133, "rewards/total_composite/std": 0.23687879741191864, "reward": 0.5034303665161133, "reward_std": 0.23687881231307983, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1071733608841896, "sampling/sampling_logp_difference/max": 1.7948522567749023, "sampling/importance_sampling_ratio/min": 0.1661520004272461, "sampling/importance_sampling_ratio/mean": 1.0025920867919922, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47142135351896286, "clip_ratio/low_mean": 0.030039405450224876, "clip_ratio/low_min": 0.030039405450224876, "clip_ratio/high_mean": 0.06009481102228165, "clip_ratio/high_max": 0.06009481102228165, "clip_ratio/region_mean": 0.09013421647250652, "reward_total_mean": 0.5034303665161133, "reward_meter_mean": 0.4706726372241974, "reward_meter_std": 0.3150758743286133, "reward_count_adherence_mean": 0.8541666269302368, "reward_count_adherence_std": 0.3500283658504486, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8487769961357117, "reward_repeat_soft_std": 0.11049899458885193, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.1345893144607544, "reward_total_composite_mean": 0.5034303665161133, "reward_total_composite_std": 0.23687879741191864} {"timestamp_utc": "2026-04-13T04:01:28Z", "mode": "train", "global_step": 2270, "epoch": 0.22802611752887997, "loss": -0.0169, "grad_norm": 12.392192840576172, "learning_rate": 3.1242424242424245e-06, "num_tokens": 4174874.0, "completions/mean_length": 54.25, "completions/min_length": 38.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.25, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.3040558099746704, "rewards/meter/std": 0.37182384729385376, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9944364428520203, "rewards/repeat_soft/std": 0.00841290783137083, "rewards/judge_quality/mean": 0.5575000047683716, "rewards/judge_quality/std": 0.19955310225486755, "rewards/total_composite/mean": 0.5535187721252441, "rewards/total_composite/std": 0.2203553467988968, "reward": 0.5535187721252441, "reward_std": 0.2203553467988968, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1481311321258545, "sampling/sampling_logp_difference/max": 2.317870616912842, "sampling/importance_sampling_ratio/min": 0.09848307818174362, "sampling/importance_sampling_ratio/mean": 1.0037918090820312, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6278055943548679, "clip_ratio/low_mean": 0.07778972154483199, "clip_ratio/low_min": 0.07778972154483199, "clip_ratio/high_mean": 0.047908693086355925, "clip_ratio/high_max": 0.047908693086355925, "clip_ratio/region_mean": 0.12569841463118792, "reward_total_mean": 0.5535187721252441, "reward_meter_mean": 0.3040558099746704, "reward_meter_std": 0.37182384729385376, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9944364428520203, "reward_repeat_soft_std": 0.00841290783137083, "reward_judge_quality_mean": 0.5575000047683716, "reward_judge_quality_std": 0.19955310225486755, "reward_total_composite_mean": 0.5535187721252441, "reward_total_composite_std": 0.2203553467988968} {"timestamp_utc": "2026-04-13T04:01:35Z", "mode": "train", "global_step": 2271, "epoch": 0.22812656956303365, "loss": -0.0069, "grad_norm": 12.415712356567383, "learning_rate": 3.1212121212121217e-06, "num_tokens": 4176639.0, "completions/mean_length": 67.625, "completions/min_length": 62.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.625, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.6960297226905823, "rewards/meter/std": 0.35723763704299927, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9948195219039917, "rewards/repeat_soft/std": 0.009195586666464806, "rewards/judge_quality/mean": 0.45625001192092896, "rewards/judge_quality/std": 0.308032363653183, "rewards/total_composite/mean": 0.6995702981948853, "rewards/total_composite/std": 0.16618075966835022, "reward": 0.6995702981948853, "reward_std": 0.16618075966835022, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1455896645784378, "sampling/sampling_logp_difference/max": 1.650446891784668, "sampling/importance_sampling_ratio/min": 0.19196410477161407, "sampling/importance_sampling_ratio/mean": 1.025964379310608, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9125097692012787, "clip_ratio/low_mean": 0.05721929203718901, "clip_ratio/low_min": 0.05721929203718901, "clip_ratio/high_mean": 0.07524772267788649, "clip_ratio/high_max": 0.07524772267788649, "clip_ratio/region_mean": 0.1324670147150755, "reward_total_mean": 0.6995702981948853, "reward_meter_mean": 0.6960297226905823, "reward_meter_std": 0.35723763704299927, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9948195219039917, "reward_repeat_soft_std": 0.009195586666464806, "reward_judge_quality_mean": 0.45625001192092896, "reward_judge_quality_std": 0.308032363653183, "reward_total_composite_mean": 0.6995702981948853, "reward_total_composite_std": 0.16618075966835022} {"timestamp_utc": "2026-04-13T04:01:44Z", "mode": "train", "global_step": 2272, "epoch": 0.22822702159718733, "loss": 0.0519, "grad_norm": 6.6104512214660645, "learning_rate": 3.1181818181818186e-06, "num_tokens": 4179589.0, "completions/mean_length": 170.75, "completions/min_length": 152.0, "completions/max_length": 178.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 170.75, "completions/min_terminated_length": 152.0, "completions/max_terminated_length": 178.0, "rewards/meter/mean": 0.9921021461486816, "rewards/meter/std": 0.0042611719109117985, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9671103954315186, "rewards/repeat_soft/std": 0.013648624531924725, "rewards/judge_quality/mean": 0.5112500190734863, "rewards/judge_quality/std": 0.18216457962989807, "rewards/total_composite/mean": 0.8465319871902466, "rewards/total_composite/std": 0.05497308820486069, "reward": 0.8465319871902466, "reward_std": 0.05497307702898979, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10946008563041687, "sampling/sampling_logp_difference/max": 2.7567484378814697, "sampling/importance_sampling_ratio/min": 0.06349790096282959, "sampling/importance_sampling_ratio/mean": 1.0073812007904053, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5568156726658344, "clip_ratio/low_mean": 0.052405111491680145, "clip_ratio/low_min": 0.052405111491680145, "clip_ratio/high_mean": 0.03606364969164133, "clip_ratio/high_max": 0.03606364969164133, "clip_ratio/region_mean": 0.08846876118332148, "reward_total_mean": 0.8465319871902466, "reward_meter_mean": 0.9921021461486816, "reward_meter_std": 0.0042611719109117985, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9671103954315186, "reward_repeat_soft_std": 0.013648624531924725, "reward_judge_quality_mean": 0.5112500190734863, "reward_judge_quality_std": 0.18216457962989807, "reward_total_composite_mean": 0.8465319871902466, "reward_total_composite_std": 0.05497308820486069} {"timestamp_utc": "2026-04-13T04:01:51Z", "mode": "train", "global_step": 2273, "epoch": 0.22832747363134104, "loss": 0.0516, "grad_norm": 7.1361284255981445, "learning_rate": 3.1151515151515154e-06, "num_tokens": 4181569.0, "completions/mean_length": 95.5, "completions/min_length": 83.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.5, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.7144849896430969, "rewards/meter/std": 0.3136322498321533, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9907899498939514, "rewards/repeat_soft/std": 0.007143169641494751, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.13265827298164368, "rewards/total_composite/mean": 0.7224721908569336, "rewards/total_composite/std": 0.1345498114824295, "reward": 0.7224721908569336, "reward_std": 0.1345497965812683, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13473881781101227, "sampling/sampling_logp_difference/max": 1.4409198760986328, "sampling/importance_sampling_ratio/min": 0.23670992255210876, "sampling/importance_sampling_ratio/mean": 1.0018792152404785, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7919429764151573, "clip_ratio/low_mean": 0.048998682759702206, "clip_ratio/low_min": 0.048998682759702206, "clip_ratio/high_mean": 0.10037583578377962, "clip_ratio/high_max": 0.10037583578377962, "clip_ratio/region_mean": 0.14937451854348183, "reward_total_mean": 0.7224721908569336, "reward_meter_mean": 0.7144849896430969, "reward_meter_std": 0.3136322498321533, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9907899498939514, "reward_repeat_soft_std": 0.007143169641494751, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.13265827298164368, "reward_total_composite_mean": 0.7224721908569336, "reward_total_composite_std": 0.1345498114824295} {"timestamp_utc": "2026-04-13T04:01:59Z", "mode": "train", "global_step": 2274, "epoch": 0.22842792566549472, "loss": 0.0153, "grad_norm": 8.691330909729004, "learning_rate": 3.1121212121212126e-06, "num_tokens": 4183804.0, "completions/mean_length": 98.375, "completions/min_length": 83.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.375, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.8796262145042419, "rewards/meter/std": 0.08525096625089645, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9041081666946411, "rewards/repeat_soft/std": 0.062280070036649704, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7419925928115845, "rewards/total_composite/std": 0.06013254448771477, "reward": 0.7419925928115845, "reward_std": 0.06013254448771477, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11264295130968094, "sampling/sampling_logp_difference/max": 2.1208629608154297, "sampling/importance_sampling_ratio/min": 0.11992809176445007, "sampling/importance_sampling_ratio/mean": 0.9946390986442566, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4574294537305832, "clip_ratio/low_mean": 0.016917502973228693, "clip_ratio/low_min": 0.016917502973228693, "clip_ratio/high_mean": 0.07684421259909868, "clip_ratio/high_max": 0.07684421259909868, "clip_ratio/region_mean": 0.09376171557232738, "reward_total_mean": 0.7419925928115845, "reward_meter_mean": 0.8796262145042419, "reward_meter_std": 0.08525096625089645, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9041081666946411, "reward_repeat_soft_std": 0.062280070036649704, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7419925928115845, "reward_total_composite_std": 0.06013254448771477} {"timestamp_utc": "2026-04-13T04:02:06Z", "mode": "train", "global_step": 2275, "epoch": 0.22852837769964843, "loss": 0.0176, "grad_norm": 16.079824447631836, "learning_rate": 3.1090909090909095e-06, "num_tokens": 4185442.0, "completions/mean_length": 54.75, "completions/min_length": 51.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.75, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9769181609153748, "rewards/meter/std": 0.030756711959838867, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.96026611328125, "rewards/repeat_soft/std": 0.018429752439260483, "rewards/judge_quality/mean": 0.45124998688697815, "rewards/judge_quality/std": 0.12799973785877228, "rewards/total_composite/mean": 0.8210147619247437, "rewards/total_composite/std": 0.041720304638147354, "reward": 0.8210147619247437, "reward_std": 0.041720300912857056, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1070127859711647, "sampling/sampling_logp_difference/max": 2.958751916885376, "sampling/importance_sampling_ratio/min": 0.05188363417983055, "sampling/importance_sampling_ratio/mean": 0.9967778325080872, "sampling/importance_sampling_ratio/max": 1.807612657546997, "entropy": 0.5000924579799175, "clip_ratio/low_mean": 0.04866687906906009, "clip_ratio/low_min": 0.04866687906906009, "clip_ratio/high_mean": 0.05049362452700734, "clip_ratio/high_max": 0.05049362452700734, "clip_ratio/region_mean": 0.09916050359606743, "reward_total_mean": 0.8210147619247437, "reward_meter_mean": 0.9769181609153748, "reward_meter_std": 0.030756711959838867, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.96026611328125, "reward_repeat_soft_std": 0.018429752439260483, "reward_judge_quality_mean": 0.45124998688697815, "reward_judge_quality_std": 0.12799973785877228, "reward_total_composite_mean": 0.8210147619247437, "reward_total_composite_std": 0.041720304638147354} {"timestamp_utc": "2026-04-13T04:02:15Z", "mode": "train", "global_step": 2276, "epoch": 0.2286288297338021, "loss": -0.0192, "grad_norm": 7.701524257659912, "learning_rate": 3.1060606060606063e-06, "num_tokens": 4188232.0, "completions/mean_length": 157.75, "completions/min_length": 138.0, "completions/max_length": 173.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 157.75, "completions/min_terminated_length": 138.0, "completions/max_terminated_length": 173.0, "rewards/meter/mean": 0.8909463286399841, "rewards/meter/std": 0.16674642264842987, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9882433414459229, "rewards/repeat_soft/std": 0.009209093637764454, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.769375205039978, "rewards/total_composite/std": 0.09230349212884903, "reward": 0.769375205039978, "reward_std": 0.09230349957942963, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12384030222892761, "sampling/sampling_logp_difference/max": 3.6565358638763428, "sampling/importance_sampling_ratio/min": 0.025821808725595474, "sampling/importance_sampling_ratio/mean": 1.0135295391082764, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6987558305263519, "clip_ratio/low_mean": 0.013586956076323986, "clip_ratio/low_min": 0.013586956076323986, "clip_ratio/high_mean": 0.10963591653853655, "clip_ratio/high_max": 0.10963591653853655, "clip_ratio/region_mean": 0.12322287261486053, "reward_total_mean": 0.769375205039978, "reward_meter_mean": 0.8909463286399841, "reward_meter_std": 0.16674642264842987, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9882433414459229, "reward_repeat_soft_std": 0.009209093637764454, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.769375205039978, "reward_total_composite_std": 0.09230349212884903} {"timestamp_utc": "2026-04-13T04:02:21Z", "mode": "train", "global_step": 2277, "epoch": 0.2287292817679558, "loss": -0.0048, "grad_norm": 14.255769729614258, "learning_rate": 3.103030303030303e-06, "num_tokens": 4190247.0, "completions/mean_length": 57.875, "completions/min_length": 53.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.875, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.7885715961456299, "rewards/meter/std": 0.29477909207344055, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9736809730529785, "rewards/repeat_soft/std": 0.018402019515633583, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7282252907752991, "rewards/total_composite/std": 0.13253530859947205, "reward": 0.7282252907752991, "reward_std": 0.13253530859947205, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14169473946094513, "sampling/sampling_logp_difference/max": 3.8278748989105225, "sampling/importance_sampling_ratio/min": 0.021755799651145935, "sampling/importance_sampling_ratio/mean": 0.9849969744682312, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.482583474367857, "clip_ratio/low_mean": 0.022922873497009277, "clip_ratio/low_min": 0.022922873497009277, "clip_ratio/high_mean": 0.10380907356739044, "clip_ratio/high_max": 0.10380907356739044, "clip_ratio/region_mean": 0.12673194706439972, "reward_total_mean": 0.7282252907752991, "reward_meter_mean": 0.7885715961456299, "reward_meter_std": 0.29477909207344055, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9736809730529785, "reward_repeat_soft_std": 0.018402019515633583, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7282252907752991, "reward_total_composite_std": 0.13253530859947205} {"timestamp_utc": "2026-04-13T04:02:27Z", "mode": "train", "global_step": 2278, "epoch": 0.2288297338021095, "loss": 0.0053, "grad_norm": 10.30069637298584, "learning_rate": 3.1000000000000004e-06, "num_tokens": 4191934.0, "completions/mean_length": 50.875, "completions/min_length": 48.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.875, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9769315123558044, "rewards/meter/std": 0.020798571407794952, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9668515920639038, "rewards/repeat_soft/std": 0.018220752477645874, "rewards/judge_quality/mean": 0.6237499713897705, "rewards/judge_quality/std": 0.22890658676624298, "rewards/total_composite/mean": 0.8734292984008789, "rewards/total_composite/std": 0.06511781364679337, "reward": 0.8734292984008789, "reward_std": 0.06511780619621277, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10448219627141953, "sampling/sampling_logp_difference/max": 1.5020720958709717, "sampling/importance_sampling_ratio/min": 0.22266829013824463, "sampling/importance_sampling_ratio/mean": 1.0278923511505127, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6696077957749367, "clip_ratio/low_mean": 0.055445075035095215, "clip_ratio/low_min": 0.055445075035095215, "clip_ratio/high_mean": 0.060619658790528774, "clip_ratio/high_max": 0.060619658790528774, "clip_ratio/region_mean": 0.11606473382562399, "reward_total_mean": 0.8734292984008789, "reward_meter_mean": 0.9769315123558044, "reward_meter_std": 0.020798571407794952, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9668515920639038, "reward_repeat_soft_std": 0.018220752477645874, "reward_judge_quality_mean": 0.6237499713897705, "reward_judge_quality_std": 0.22890658676624298, "reward_total_composite_mean": 0.8734292984008789, "reward_total_composite_std": 0.06511781364679337} {"timestamp_utc": "2026-04-13T04:02:39Z", "mode": "train", "global_step": 2279, "epoch": 0.22893018583626318, "loss": -0.1833, "grad_norm": 2.005967617034912, "learning_rate": 3.0969696969696972e-06, "num_tokens": 4193828.0, "completions/mean_length": 145.75, "completions/min_length": 87.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 93.42857360839844, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.9048057794570923, "rewards/meter/std": 0.17669163644313812, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9549049139022827, "rewards/repeat_soft/std": 0.02705976366996765, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.22797243297100067, "rewards/total_composite/mean": 0.723576545715332, "rewards/total_composite/std": 0.2979421615600586, "reward": 0.723576545715332, "reward_std": 0.2979421615600586, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1129932701587677, "sampling/sampling_logp_difference/max": 1.7037038803100586, "sampling/importance_sampling_ratio/min": 0.18200813233852386, "sampling/importance_sampling_ratio/mean": 0.9996045231819153, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48585284873843193, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09907463286072016, "clip_ratio/high_max": 0.09907463286072016, "clip_ratio/region_mean": 0.09907463286072016, "reward_total_mean": 0.723576545715332, "reward_meter_mean": 0.9048057794570923, "reward_meter_std": 0.17669163644313812, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9549049139022827, "reward_repeat_soft_std": 0.02705976366996765, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.22797243297100067, "reward_total_composite_mean": 0.723576545715332, "reward_total_composite_std": 0.2979421615600586} {"timestamp_utc": "2026-04-13T04:02:46Z", "mode": "train", "global_step": 2280, "epoch": 0.2290306378704169, "loss": -0.0561, "grad_norm": 7.936528205871582, "learning_rate": 3.093939393939394e-06, "num_tokens": 4195549.0, "completions/mean_length": 64.125, "completions/min_length": 53.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.125, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.98819899559021, "rewards/meter/std": 0.014654062688350677, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9720020294189453, "rewards/repeat_soft/std": 0.013057850301265717, "rewards/judge_quality/mean": 0.4024999737739563, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.8126397728919983, "rewards/total_composite/std": 0.024405276402831078, "reward": 0.8126397728919983, "reward_std": 0.024405265226960182, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12503854930400848, "sampling/sampling_logp_difference/max": 1.751694679260254, "sampling/importance_sampling_ratio/min": 0.17347970604896545, "sampling/importance_sampling_ratio/mean": 1.0067580938339233, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7550717815756798, "clip_ratio/low_mean": 0.01179245300590992, "clip_ratio/low_min": 0.01179245300590992, "clip_ratio/high_mean": 0.10649511590600014, "clip_ratio/high_max": 0.10649511590600014, "clip_ratio/region_mean": 0.11828756891191006, "reward_total_mean": 0.8126397728919983, "reward_meter_mean": 0.98819899559021, "reward_meter_std": 0.014654062688350677, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9720020294189453, "reward_repeat_soft_std": 0.013057850301265717, "reward_judge_quality_mean": 0.4024999737739563, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.8126397728919983, "reward_total_composite_std": 0.024405276402831078} {"timestamp_utc": "2026-04-13T04:02:55Z", "mode": "train", "global_step": 2281, "epoch": 0.22913108990457057, "loss": 0.0229, "grad_norm": 5.561137676239014, "learning_rate": 3.090909090909091e-06, "num_tokens": 4198319.0, "completions/mean_length": 169.25, "completions/min_length": 152.0, "completions/max_length": 185.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 169.25, "completions/min_terminated_length": 152.0, "completions/max_terminated_length": 185.0, "rewards/meter/mean": 0.9894179105758667, "rewards/meter/std": 0.004995398689061403, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9529138803482056, "rewards/repeat_soft/std": 0.03070123679935932, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.13866712152957916, "rewards/total_composite/mean": 0.8075294494628906, "rewards/total_composite/std": 0.04397060349583626, "reward": 0.8075294494628906, "reward_std": 0.043970607221126556, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10710109025239944, "sampling/sampling_logp_difference/max": 2.3992815017700195, "sampling/importance_sampling_ratio/min": 0.09078315645456314, "sampling/importance_sampling_ratio/mean": 1.0131112337112427, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7298790365457535, "clip_ratio/low_mean": 0.019864341244101524, "clip_ratio/low_min": 0.019864341244101524, "clip_ratio/high_mean": 0.06906153820455074, "clip_ratio/high_max": 0.06906153820455074, "clip_ratio/region_mean": 0.08892587944865227, "reward_total_mean": 0.8075294494628906, "reward_meter_mean": 0.9894179105758667, "reward_meter_std": 0.004995398689061403, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9529138803482056, "reward_repeat_soft_std": 0.03070123679935932, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.13866712152957916, "reward_total_composite_mean": 0.8075294494628906, "reward_total_composite_std": 0.04397060349583626} {"timestamp_utc": "2026-04-13T04:03:03Z", "mode": "train", "global_step": 2282, "epoch": 0.22923154193872425, "loss": 0.1129, "grad_norm": 11.197205543518066, "learning_rate": 3.087878787878788e-06, "num_tokens": 4200201.0, "completions/mean_length": 65.25, "completions/min_length": 59.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.25, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.9853428602218628, "rewards/meter/std": 0.01178208738565445, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9689719676971436, "rewards/repeat_soft/std": 0.04257161542773247, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.8110514879226685, "rewards/total_composite/std": 0.022287622094154358, "reward": 0.8110514879226685, "reward_std": 0.022287609055638313, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11657815426588058, "sampling/sampling_logp_difference/max": 1.9435815811157227, "sampling/importance_sampling_ratio/min": 0.1431901752948761, "sampling/importance_sampling_ratio/mean": 1.002750277519226, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.726557120680809, "clip_ratio/low_mean": 0.019004065543413162, "clip_ratio/low_min": 0.019004065543413162, "clip_ratio/high_mean": 0.07935759983956814, "clip_ratio/high_max": 0.07935759983956814, "clip_ratio/region_mean": 0.0983616653829813, "reward_total_mean": 0.8110514879226685, "reward_meter_mean": 0.9853428602218628, "reward_meter_std": 0.01178208738565445, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9689719676971436, "reward_repeat_soft_std": 0.04257161542773247, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.8110514879226685, "reward_total_composite_std": 0.022287622094154358} {"timestamp_utc": "2026-04-13T04:03:11Z", "mode": "train", "global_step": 2283, "epoch": 0.22933199397287796, "loss": 0.0136, "grad_norm": 5.543764591217041, "learning_rate": 3.084848484848485e-06, "num_tokens": 4203183.0, "completions/mean_length": 158.75, "completions/min_length": 148.0, "completions/max_length": 183.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 158.75, "completions/min_terminated_length": 148.0, "completions/max_terminated_length": 183.0, "rewards/meter/mean": 0.9931119680404663, "rewards/meter/std": 0.004816501401364803, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9254305362701416, "rewards/repeat_soft/std": 0.046441756188869476, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.8026934266090393, "rewards/total_composite/std": 0.024190984666347504, "reward": 0.8026934266090393, "reward_std": 0.02419096790254116, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09599343687295914, "sampling/sampling_logp_difference/max": 2.6016573905944824, "sampling/importance_sampling_ratio/min": 0.07415057718753815, "sampling/importance_sampling_ratio/mean": 1.003470540046692, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4753006473183632, "clip_ratio/low_mean": 0.025672225281596184, "clip_ratio/low_min": 0.025672225281596184, "clip_ratio/high_mean": 0.06656351033598185, "clip_ratio/high_max": 0.06656351033598185, "clip_ratio/region_mean": 0.09223573561757803, "reward_total_mean": 0.8026934266090393, "reward_meter_mean": 0.9931119680404663, "reward_meter_std": 0.004816501401364803, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9254305362701416, "reward_repeat_soft_std": 0.046441756188869476, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.8026934266090393, "reward_total_composite_std": 0.024190984666347504} {"timestamp_utc": "2026-04-13T04:03:19Z", "mode": "train", "global_step": 2284, "epoch": 0.22943244600703164, "loss": 0.0249, "grad_norm": 6.481086730957031, "learning_rate": 3.081818181818182e-06, "num_tokens": 4205667.0, "completions/mean_length": 122.5, "completions/min_length": 109.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.5, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.9802255630493164, "rewards/meter/std": 0.007397334091365337, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9113805294036865, "rewards/repeat_soft/std": 0.055093344300985336, "rewards/judge_quality/mean": 0.30124998092651367, "rewards/judge_quality/std": 0.10398317128419876, "rewards/total_composite/mean": 0.772614598274231, "rewards/total_composite/std": 0.03212083876132965, "reward": 0.772614598274231, "reward_std": 0.03212084621191025, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09257900714874268, "sampling/sampling_logp_difference/max": 2.034254312515259, "sampling/importance_sampling_ratio/min": 0.130777969956398, "sampling/importance_sampling_ratio/mean": 1.0033923387527466, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5369868688285351, "clip_ratio/low_mean": 0.05839605536311865, "clip_ratio/low_min": 0.05839605536311865, "clip_ratio/high_mean": 0.03840387146919966, "clip_ratio/high_max": 0.03840387146919966, "clip_ratio/region_mean": 0.0967999268323183, "reward_total_mean": 0.772614598274231, "reward_meter_mean": 0.9802255630493164, "reward_meter_std": 0.007397334091365337, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9113805294036865, "reward_repeat_soft_std": 0.055093344300985336, "reward_judge_quality_mean": 0.30124998092651367, "reward_judge_quality_std": 0.10398317128419876, "reward_total_composite_mean": 0.772614598274231, "reward_total_composite_std": 0.03212083876132965} {"timestamp_utc": "2026-04-13T04:03:27Z", "mode": "train", "global_step": 2285, "epoch": 0.22953289804118535, "loss": 0.0393, "grad_norm": 6.832182884216309, "learning_rate": 3.078787878787879e-06, "num_tokens": 4208021.0, "completions/mean_length": 116.25, "completions/min_length": 100.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.25, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.9032461047172546, "rewards/meter/std": 0.2341921180486679, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9112030267715454, "rewards/repeat_soft/std": 0.04052918404340744, "rewards/judge_quality/mean": 0.3187499940395355, "rewards/judge_quality/std": 0.1397382616996765, "rewards/total_composite/mean": 0.7432060241699219, "rewards/total_composite/std": 0.10118047147989273, "reward": 0.7432060241699219, "reward_std": 0.10118046402931213, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09130173921585083, "sampling/sampling_logp_difference/max": 1.939577579498291, "sampling/importance_sampling_ratio/min": 0.1437646746635437, "sampling/importance_sampling_ratio/mean": 1.0016119480133057, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5038419291377068, "clip_ratio/low_mean": 0.03774048946797848, "clip_ratio/low_min": 0.03774048946797848, "clip_ratio/high_mean": 0.044297599233686924, "clip_ratio/high_max": 0.044297599233686924, "clip_ratio/region_mean": 0.0820380887016654, "reward_total_mean": 0.7432060241699219, "reward_meter_mean": 0.9032461047172546, "reward_meter_std": 0.2341921180486679, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9112030267715454, "reward_repeat_soft_std": 0.04052918404340744, "reward_judge_quality_mean": 0.3187499940395355, "reward_judge_quality_std": 0.1397382616996765, "reward_total_composite_mean": 0.7432060241699219, "reward_total_composite_std": 0.10118047147989273} {"timestamp_utc": "2026-04-13T04:03:33Z", "mode": "train", "global_step": 2286, "epoch": 0.22963335007533903, "loss": 0.035, "grad_norm": 9.696159362792969, "learning_rate": 3.075757575757576e-06, "num_tokens": 4209615.0, "completions/mean_length": 55.25, "completions/min_length": 48.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.25, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9770018458366394, "rewards/meter/std": 0.014669987373054028, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624063372612, "rewards/repeat_soft/std": 0.029023559764027596, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8130164742469788, "rewards/total_composite/std": 0.009007977321743965, "reward": 0.8130164742469788, "reward_std": 0.00900797639042139, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10904061794281006, "sampling/sampling_logp_difference/max": 1.42055082321167, "sampling/importance_sampling_ratio/min": 0.24158091843128204, "sampling/importance_sampling_ratio/mean": 1.0033782720565796, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7606367245316505, "clip_ratio/low_mean": 0.02574732154607773, "clip_ratio/low_min": 0.02574732154607773, "clip_ratio/high_mean": 0.061366185545921326, "clip_ratio/high_max": 0.061366185545921326, "clip_ratio/region_mean": 0.08711350709199905, "reward_total_mean": 0.8130164742469788, "reward_meter_mean": 0.9770018458366394, "reward_meter_std": 0.014669987373054028, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624063372612, "reward_repeat_soft_std": 0.029023559764027596, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8130164742469788, "reward_total_composite_std": 0.009007977321743965} {"timestamp_utc": "2026-04-13T04:03:40Z", "mode": "train", "global_step": 2287, "epoch": 0.2297338021094927, "loss": 0.0248, "grad_norm": 9.403878211975098, "learning_rate": 3.0727272727272727e-06, "num_tokens": 4211749.0, "completions/mean_length": 94.75, "completions/min_length": 83.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.75, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.8874468803405762, "rewards/meter/std": 0.160160094499588, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9511683583259583, "rewards/repeat_soft/std": 0.04611232131719589, "rewards/judge_quality/mean": 0.35999998450279236, "rewards/judge_quality/std": 0.09165150672197342, "rewards/total_composite/mean": 0.752467930316925, "rewards/total_composite/std": 0.07006832957267761, "reward": 0.752467930316925, "reward_std": 0.0700683444738388, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15254180133342743, "sampling/sampling_logp_difference/max": 1.905494213104248, "sampling/importance_sampling_ratio/min": 0.14874911308288574, "sampling/importance_sampling_ratio/mean": 1.0030423402786255, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7656541913747787, "clip_ratio/low_mean": 0.03523102402687073, "clip_ratio/low_min": 0.03523102402687073, "clip_ratio/high_mean": 0.10495825204998255, "clip_ratio/high_max": 0.10495825204998255, "clip_ratio/region_mean": 0.14018927607685328, "reward_total_mean": 0.752467930316925, "reward_meter_mean": 0.8874468803405762, "reward_meter_std": 0.160160094499588, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9511683583259583, "reward_repeat_soft_std": 0.04611232131719589, "reward_judge_quality_mean": 0.35999998450279236, "reward_judge_quality_std": 0.09165150672197342, "reward_total_composite_mean": 0.752467930316925, "reward_total_composite_std": 0.07006832957267761} {"timestamp_utc": "2026-04-13T04:03:49Z", "mode": "train", "global_step": 2288, "epoch": 0.22983425414364642, "loss": 0.0524, "grad_norm": 6.272814750671387, "learning_rate": 3.0696969696969696e-06, "num_tokens": 4214822.0, "completions/mean_length": 200.125, "completions/min_length": 175.0, "completions/max_length": 241.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 200.125, "completions/min_terminated_length": 175.0, "completions/max_terminated_length": 241.0, "rewards/meter/mean": 0.7831336259841919, "rewards/meter/std": 0.22752954065799713, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9354701042175293, "rewards/repeat_soft/std": 0.037795379757881165, "rewards/judge_quality/mean": 0.3725000023841858, "rewards/judge_quality/std": 0.2494422346353531, "rewards/total_composite/mean": 0.701457142829895, "rewards/total_composite/std": 0.1117960661649704, "reward": 0.701457142829895, "reward_std": 0.1117960587143898, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10932788252830505, "sampling/sampling_logp_difference/max": 2.84452223777771, "sampling/importance_sampling_ratio/min": 0.05816204845905304, "sampling/importance_sampling_ratio/mean": 1.0009799003601074, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6251996383070946, "clip_ratio/low_mean": 0.03063239948824048, "clip_ratio/low_min": 0.03063239948824048, "clip_ratio/high_mean": 0.07566145807504654, "clip_ratio/high_max": 0.07566145807504654, "clip_ratio/region_mean": 0.10629385756328702, "reward_total_mean": 0.701457142829895, "reward_meter_mean": 0.7831336259841919, "reward_meter_std": 0.22752954065799713, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9354701042175293, "reward_repeat_soft_std": 0.037795379757881165, "reward_judge_quality_mean": 0.3725000023841858, "reward_judge_quality_std": 0.2494422346353531, "reward_total_composite_mean": 0.701457142829895, "reward_total_composite_std": 0.1117960661649704} {"timestamp_utc": "2026-04-13T04:03:55Z", "mode": "train", "global_step": 2289, "epoch": 0.2299347061778001, "loss": 0.0666, "grad_norm": 13.925142288208008, "learning_rate": 3.066666666666667e-06, "num_tokens": 4216256.0, "completions/mean_length": 26.25, "completions/min_length": 22.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.25, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.9329569339752197, "rewards/meter/std": 0.07747205346822739, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9612975120544434, "rewards/repeat_soft/std": 0.0034009767696261406, "rewards/judge_quality/mean": 0.6225000023841858, "rewards/judge_quality/std": 0.24656209349632263, "rewards/total_composite/mean": 0.8527103662490845, "rewards/total_composite/std": 0.08204091340303421, "reward": 0.8527103662490845, "reward_std": 0.08204091340303421, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1471063196659088, "sampling/sampling_logp_difference/max": 2.4665300846099854, "sampling/importance_sampling_ratio/min": 0.08487886935472488, "sampling/importance_sampling_ratio/mean": 1.0150392055511475, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7603907883167267, "clip_ratio/low_mean": 0.04945970978587866, "clip_ratio/low_min": 0.04945970978587866, "clip_ratio/high_mean": 0.06009615398943424, "clip_ratio/high_max": 0.06009615398943424, "clip_ratio/region_mean": 0.1095558637753129, "reward_total_mean": 0.8527103662490845, "reward_meter_mean": 0.9329569339752197, "reward_meter_std": 0.07747205346822739, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9612975120544434, "reward_repeat_soft_std": 0.0034009767696261406, "reward_judge_quality_mean": 0.6225000023841858, "reward_judge_quality_std": 0.24656209349632263, "reward_total_composite_mean": 0.8527103662490845, "reward_total_composite_std": 0.08204091340303421} {"timestamp_utc": "2026-04-13T04:04:02Z", "mode": "train", "global_step": 2290, "epoch": 0.2300351582119538, "loss": 0.0151, "grad_norm": 4.518292427062988, "learning_rate": 3.0636363636363636e-06, "num_tokens": 4218625.0, "completions/mean_length": 120.125, "completions/min_length": 115.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.125, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.9738433361053467, "rewards/meter/std": 0.0272081159055233, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7897461652755737, "rewards/repeat_soft/std": 0.06155554950237274, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.8307040929794312, "rewards/total_composite/std": 0.06945384293794632, "reward": 0.8307040929794312, "reward_std": 0.06945384293794632, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06942243129014969, "sampling/sampling_logp_difference/max": 1.5225679874420166, "sampling/importance_sampling_ratio/min": 0.21815097332000732, "sampling/importance_sampling_ratio/mean": 1.013007402420044, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39358795434236526, "clip_ratio/low_mean": 0.052003593649715185, "clip_ratio/low_min": 0.052003593649715185, "clip_ratio/high_mean": 0.019943268969655037, "clip_ratio/high_max": 0.019943268969655037, "clip_ratio/region_mean": 0.07194686261937022, "reward_total_mean": 0.8307040929794312, "reward_meter_mean": 0.9738433361053467, "reward_meter_std": 0.0272081159055233, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7897461652755737, "reward_repeat_soft_std": 0.06155554950237274, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.8307040929794312, "reward_total_composite_std": 0.06945384293794632} {"timestamp_utc": "2026-04-13T04:04:09Z", "mode": "train", "global_step": 2291, "epoch": 0.23013561024610749, "loss": 0.0376, "grad_norm": 7.582006454467773, "learning_rate": 3.0606060606060605e-06, "num_tokens": 4221039.0, "completions/mean_length": 125.75, "completions/min_length": 117.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.75, "completions/min_terminated_length": 117.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.7402828931808472, "rewards/meter/std": 0.16870002448558807, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9630497694015503, "rewards/repeat_soft/std": 0.017146261408925056, "rewards/judge_quality/mean": 0.3050000071525574, "rewards/judge_quality/std": 0.10928337275981903, "rewards/total_composite/mean": 0.6709322929382324, "rewards/total_composite/std": 0.07568971067667007, "reward": 0.6709322929382324, "reward_std": 0.07568971812725067, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1334531158208847, "sampling/sampling_logp_difference/max": 2.7418715953826904, "sampling/importance_sampling_ratio/min": 0.06444960832595825, "sampling/importance_sampling_ratio/mean": 1.0086894035339355, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8221222683787346, "clip_ratio/low_mean": 0.08307115733623505, "clip_ratio/low_min": 0.08307115733623505, "clip_ratio/high_mean": 0.05315290205180645, "clip_ratio/high_max": 0.05315290205180645, "clip_ratio/region_mean": 0.1362240593880415, "reward_total_mean": 0.6709322929382324, "reward_meter_mean": 0.7402828931808472, "reward_meter_std": 0.16870002448558807, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9630497694015503, "reward_repeat_soft_std": 0.017146261408925056, "reward_judge_quality_mean": 0.3050000071525574, "reward_judge_quality_std": 0.10928337275981903, "reward_total_composite_mean": 0.6709322929382324, "reward_total_composite_std": 0.07568971067667007} {"timestamp_utc": "2026-04-13T04:04:17Z", "mode": "train", "global_step": 2292, "epoch": 0.23023606228026117, "loss": 0.0761, "grad_norm": 7.759284019470215, "learning_rate": 3.057575757575758e-06, "num_tokens": 4223705.0, "completions/mean_length": 143.25, "completions/min_length": 132.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 143.25, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.8958962559700012, "rewards/meter/std": 0.1533329039812088, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9967093467712402, "rewards/repeat_soft/std": 0.002091505564749241, "rewards/judge_quality/mean": 0.5724999904632568, "rewards/judge_quality/std": 0.21618445217609406, "rewards/total_composite/mean": 0.8245742321014404, "rewards/total_composite/std": 0.0916302502155304, "reward": 0.8245742321014404, "reward_std": 0.0916302353143692, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13039593398571014, "sampling/sampling_logp_difference/max": 1.638803482055664, "sampling/importance_sampling_ratio/min": 0.19421228766441345, "sampling/importance_sampling_ratio/mean": 1.0044236183166504, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.762320026755333, "clip_ratio/low_mean": 0.058176361955702305, "clip_ratio/low_min": 0.058176361955702305, "clip_ratio/high_mean": 0.05129801295697689, "clip_ratio/high_max": 0.05129801295697689, "clip_ratio/region_mean": 0.1094743749126792, "reward_total_mean": 0.8245742321014404, "reward_meter_mean": 0.8958962559700012, "reward_meter_std": 0.1533329039812088, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9967093467712402, "reward_repeat_soft_std": 0.002091505564749241, "reward_judge_quality_mean": 0.5724999904632568, "reward_judge_quality_std": 0.21618445217609406, "reward_total_composite_mean": 0.8245742321014404, "reward_total_composite_std": 0.0916302502155304} {"timestamp_utc": "2026-04-13T04:04:25Z", "mode": "train", "global_step": 2293, "epoch": 0.23033651431441488, "loss": 0.0092, "grad_norm": 15.814827919006348, "learning_rate": 3.054545454545455e-06, "num_tokens": 4225226.0, "completions/mean_length": 30.125, "completions/min_length": 24.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.125, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.5659411549568176, "rewards/meter/std": 0.40250834822654724, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.41749998927116394, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.6261735558509827, "rewards/total_composite/std": 0.1776750087738037, "reward": 0.6261735558509827, "reward_std": 0.1776750236749649, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15476317703723907, "sampling/sampling_logp_difference/max": 1.1759154796600342, "sampling/importance_sampling_ratio/min": 0.3085364103317261, "sampling/importance_sampling_ratio/mean": 1.026366949081421, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.237736962735653, "clip_ratio/low_mean": 0.04217069875448942, "clip_ratio/low_min": 0.04217069875448942, "clip_ratio/high_mean": 0.09191532153636217, "clip_ratio/high_max": 0.09191532153636217, "clip_ratio/region_mean": 0.1340860202908516, "reward_total_mean": 0.6261735558509827, "reward_meter_mean": 0.5659411549568176, "reward_meter_std": 0.40250834822654724, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.41749998927116394, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.6261735558509827, "reward_total_composite_std": 0.1776750087738037} {"timestamp_utc": "2026-04-13T04:04:37Z", "mode": "train", "global_step": 2294, "epoch": 0.23043696634856856, "loss": -0.1845, "grad_norm": 2.477029323577881, "learning_rate": 3.051515151515152e-06, "num_tokens": 4227238.0, "completions/mean_length": 145.5, "completions/min_length": 75.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 93.14286041259766, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.7873064279556274, "rewards/meter/std": 0.33043840527534485, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.17817415297031403, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9766391515731812, "rewards/repeat_soft/std": 0.01322680339217186, "rewards/judge_quality/mean": 0.4112499952316284, "rewards/judge_quality/std": 0.1797965168952942, "rewards/total_composite/mean": 0.6464303731918335, "rewards/total_composite/std": 0.29330363869667053, "reward": 0.6464303731918335, "reward_std": 0.29330363869667053, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1156885102391243, "sampling/sampling_logp_difference/max": 1.7094941139221191, "sampling/importance_sampling_ratio/min": 0.18095730245113373, "sampling/importance_sampling_ratio/mean": 1.0196905136108398, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7910239771008492, "clip_ratio/low_mean": 0.013333333656191826, "clip_ratio/low_min": 0.013333333656191826, "clip_ratio/high_mean": 0.09391502197831869, "clip_ratio/high_max": 0.09391502197831869, "clip_ratio/region_mean": 0.10724835563451052, "reward_total_mean": 0.6464303731918335, "reward_meter_mean": 0.7873064279556274, "reward_meter_std": 0.33043840527534485, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.17817415297031403, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9766391515731812, "reward_repeat_soft_std": 0.01322680339217186, "reward_judge_quality_mean": 0.4112499952316284, "reward_judge_quality_std": 0.1797965168952942, "reward_total_composite_mean": 0.6464303731918335, "reward_total_composite_std": 0.29330363869667053} {"timestamp_utc": "2026-04-13T04:04:44Z", "mode": "train", "global_step": 2295, "epoch": 0.23053741838272224, "loss": 0.0268, "grad_norm": 6.161674976348877, "learning_rate": 3.048484848484849e-06, "num_tokens": 4229422.0, "completions/mean_length": 81.0, "completions/min_length": 75.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.0, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.9480268359184265, "rewards/meter/std": 0.11726715415716171, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9352669715881348, "rewards/repeat_soft/std": 0.02530963346362114, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.8085137605667114, "rewards/total_composite/std": 0.08324775099754333, "reward": 0.8085137605667114, "reward_std": 0.08324776589870453, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0970795750617981, "sampling/sampling_logp_difference/max": 1.5510183572769165, "sampling/importance_sampling_ratio/min": 0.21203193068504333, "sampling/importance_sampling_ratio/mean": 1.0155130624771118, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6120898388326168, "clip_ratio/low_mean": 0.028129760175943375, "clip_ratio/low_min": 0.028129760175943375, "clip_ratio/high_mean": 0.08994680736213923, "clip_ratio/high_max": 0.08994680736213923, "clip_ratio/region_mean": 0.1180765675380826, "reward_total_mean": 0.8085137605667114, "reward_meter_mean": 0.9480268359184265, "reward_meter_std": 0.11726715415716171, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9352669715881348, "reward_repeat_soft_std": 0.02530963346362114, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.8085137605667114, "reward_total_composite_std": 0.08324775099754333} {"timestamp_utc": "2026-04-13T04:04:51Z", "mode": "train", "global_step": 2296, "epoch": 0.23063787041687595, "loss": 0.0313, "grad_norm": 12.437100410461426, "learning_rate": 3.045454545454546e-06, "num_tokens": 4231031.0, "completions/mean_length": 50.125, "completions/min_length": 46.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.125, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.707313060760498, "rewards/meter/std": 0.3413207232952118, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9875316619873047, "rewards/repeat_soft/std": 0.018078777939081192, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.18873640894889832, "rewards/total_composite/mean": 0.7252939939498901, "rewards/total_composite/std": 0.17764440178871155, "reward": 0.7252939939498901, "reward_std": 0.17764440178871155, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1264171302318573, "sampling/sampling_logp_difference/max": 2.4823875427246094, "sampling/importance_sampling_ratio/min": 0.08354352414608002, "sampling/importance_sampling_ratio/mean": 1.0043076276779175, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8570178970694542, "clip_ratio/low_mean": 0.035357143729925156, "clip_ratio/low_min": 0.035357143729925156, "clip_ratio/high_mean": 0.08980920258909464, "clip_ratio/high_max": 0.08980920258909464, "clip_ratio/region_mean": 0.1251663463190198, "reward_total_mean": 0.7252939939498901, "reward_meter_mean": 0.707313060760498, "reward_meter_std": 0.3413207232952118, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9875316619873047, "reward_repeat_soft_std": 0.018078777939081192, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.18873640894889832, "reward_total_composite_mean": 0.7252939939498901, "reward_total_composite_std": 0.17764440178871155} {"timestamp_utc": "2026-04-13T04:05:03Z", "mode": "train", "global_step": 2297, "epoch": 0.23073832245102963, "loss": -0.0946, "grad_norm": 2.6907942295074463, "learning_rate": 3.0424242424242427e-06, "num_tokens": 4232630.0, "completions/mean_length": 92.875, "completions/min_length": 30.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 33.0, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.8674585819244385, "rewards/meter/std": 0.2057303488254547, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9593343734741211, "rewards/repeat_soft/std": 0.008953613229095936, "rewards/judge_quality/mean": 0.4300000071525574, "rewards/judge_quality/std": 0.24454039335250854, "rewards/total_composite/mean": 0.7000468969345093, "rewards/total_composite/std": 0.30485638976097107, "reward": 0.7000468969345093, "reward_std": 0.30485638976097107, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10713575035333633, "sampling/sampling_logp_difference/max": 0.9283862113952637, "sampling/importance_sampling_ratio/min": 0.5398799777030945, "sampling/importance_sampling_ratio/mean": 1.0274230241775513, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7384084090590477, "clip_ratio/low_mean": 0.010714286006987095, "clip_ratio/low_min": 0.010714286006987095, "clip_ratio/high_mean": 0.08745502680540085, "clip_ratio/high_max": 0.08745502680540085, "clip_ratio/region_mean": 0.09816931281238794, "reward_total_mean": 0.7000468969345093, "reward_meter_mean": 0.8674585819244385, "reward_meter_std": 0.2057303488254547, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9593343734741211, "reward_repeat_soft_std": 0.008953613229095936, "reward_judge_quality_mean": 0.4300000071525574, "reward_judge_quality_std": 0.24454039335250854, "reward_total_composite_mean": 0.7000468969345093, "reward_total_composite_std": 0.30485638976097107} {"timestamp_utc": "2026-04-13T04:05:11Z", "mode": "train", "global_step": 2298, "epoch": 0.23083877448518333, "loss": 0.032, "grad_norm": 5.183167934417725, "learning_rate": 3.03939393939394e-06, "num_tokens": 4235249.0, "completions/mean_length": 160.375, "completions/min_length": 130.0, "completions/max_length": 176.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 160.375, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 176.0, "rewards/meter/mean": 0.5731065273284912, "rewards/meter/std": 0.32104089856147766, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8341456651687622, "rewards/repeat_soft/std": 0.10303957015275955, "rewards/judge_quality/mean": 0.30124998092651367, "rewards/judge_quality/std": 0.10398317128419876, "rewards/total_composite/mean": 0.5816875100135803, "rewards/total_composite/std": 0.12111090123653412, "reward": 0.5816875100135803, "reward_std": 0.12111092358827591, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09928971529006958, "sampling/sampling_logp_difference/max": 1.8840179443359375, "sampling/importance_sampling_ratio/min": 0.1519782394170761, "sampling/importance_sampling_ratio/mean": 1.0047271251678467, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6046051755547523, "clip_ratio/low_mean": 0.05107555422000587, "clip_ratio/low_min": 0.05107555422000587, "clip_ratio/high_mean": 0.050478147342801094, "clip_ratio/high_max": 0.050478147342801094, "clip_ratio/region_mean": 0.10155370156280696, "reward_total_mean": 0.5816875100135803, "reward_meter_mean": 0.5731065273284912, "reward_meter_std": 0.32104089856147766, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8341456651687622, "reward_repeat_soft_std": 0.10303957015275955, "reward_judge_quality_mean": 0.30124998092651367, "reward_judge_quality_std": 0.10398317128419876, "reward_total_composite_mean": 0.5816875100135803, "reward_total_composite_std": 0.12111090123653412} {"timestamp_utc": "2026-04-13T04:05:19Z", "mode": "train", "global_step": 2299, "epoch": 0.23093922651933702, "loss": -0.0022, "grad_norm": 8.46181869506836, "learning_rate": 3.036363636363637e-06, "num_tokens": 4237604.0, "completions/mean_length": 118.375, "completions/min_length": 107.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.375, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.9852068424224854, "rewards/meter/std": 0.007682192139327526, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9511866569519043, "rewards/repeat_soft/std": 0.035032980144023895, "rewards/judge_quality/mean": 0.2549999952316284, "rewards/judge_quality/std": 0.111867256462574, "rewards/total_composite/mean": 0.7649617195129395, "rewards/total_composite/std": 0.03583277761936188, "reward": 0.7649617195129395, "reward_std": 0.035832758992910385, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11301091313362122, "sampling/sampling_logp_difference/max": 2.6516613960266113, "sampling/importance_sampling_ratio/min": 0.07053393125534058, "sampling/importance_sampling_ratio/mean": 0.9928646683692932, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5751612931489944, "clip_ratio/low_mean": 0.05070191062986851, "clip_ratio/low_min": 0.05070191062986851, "clip_ratio/high_mean": 0.032301143277436495, "clip_ratio/high_max": 0.032301143277436495, "clip_ratio/region_mean": 0.083003053907305, "reward_total_mean": 0.7649617195129395, "reward_meter_mean": 0.9852068424224854, "reward_meter_std": 0.007682192139327526, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9511866569519043, "reward_repeat_soft_std": 0.035032980144023895, "reward_judge_quality_mean": 0.2549999952316284, "reward_judge_quality_std": 0.111867256462574, "reward_total_composite_mean": 0.7649617195129395, "reward_total_composite_std": 0.03583277761936188} {"timestamp_utc": "2026-04-13T04:05:27Z", "mode": "train", "global_step": 2300, "epoch": 0.2310396785534907, "loss": -0.0115, "grad_norm": 5.202208995819092, "learning_rate": 3.0333333333333337e-06, "num_tokens": 4240019.0, "completions/mean_length": 133.875, "completions/min_length": 121.0, "completions/max_length": 150.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 133.875, "completions/min_terminated_length": 121.0, "completions/max_terminated_length": 150.0, "rewards/meter/mean": 0.9048528075218201, "rewards/meter/std": 0.16611260175704956, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9386783838272095, "rewards/repeat_soft/std": 0.04411354660987854, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7643016576766968, "rewards/total_composite/std": 0.07016021013259888, "reward": 0.7643016576766968, "reward_std": 0.07016020268201828, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10767704993486404, "sampling/sampling_logp_difference/max": 1.980180263519287, "sampling/importance_sampling_ratio/min": 0.1380443572998047, "sampling/importance_sampling_ratio/mean": 1.0189894437789917, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6933839321136475, "clip_ratio/low_mean": 0.022842801176011562, "clip_ratio/low_min": 0.022842801176011562, "clip_ratio/high_mean": 0.07209552731364965, "clip_ratio/high_max": 0.07209552731364965, "clip_ratio/region_mean": 0.09493832848966122, "reward_total_mean": 0.7643016576766968, "reward_meter_mean": 0.9048528075218201, "reward_meter_std": 0.16611260175704956, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9386783838272095, "reward_repeat_soft_std": 0.04411354660987854, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7643016576766968, "reward_total_composite_std": 0.07016021013259888} {"timestamp_utc": "2026-04-13T04:06:35Z", "mode": "eval", "global_step": 2300, "epoch": 0.2310396785534907, "eval_loss": NaN, "eval_runtime": 67.557, "eval_samples_per_second": 1.184, "eval_steps_per_second": 0.148, "eval_num_tokens": 4240019.0, "eval_completions/mean_length": 110.9875, "eval_completions/min_length": 40.3, "eval_completions/max_length": 266.4, "eval_completions/clipped_ratio": 0.0375, "eval_completions/mean_terminated_length": 95.55178680419922, "eval_completions/min_terminated_length": 40.3, "eval_completions/max_terminated_length": 165.5, "eval_rewards/meter/mean": 0.8406145215034485, "eval_rewards/meter/std": 0.21775756597053259, "eval_rewards/count_adherence/mean": 0.9806249916553498, "eval_rewards/count_adherence/std": 0.049967176467180255, "eval_rewards/hard_gate/mean": 0.9625, "eval_rewards/hard_gate/std": 0.10606601536273956, "eval_rewards/repeat_soft/mean": 0.9468575239181518, "eval_rewards/repeat_soft/std": 0.05658040288835764, "eval_rewards/judge_quality/mean": 0.36937499046325684, "eval_rewards/judge_quality/std": 0.1088132050819695, "eval_rewards/total_composite/mean": 0.7133846759796143, "eval_rewards/total_composite/std": 0.1489717047661543, "eval_reward": 0.7133846759796143, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.06067037172615528, "eval_sampling/sampling_logp_difference/max": 1.015919828414917, "eval_sampling/importance_sampling_ratio/min": 0.3724127531051636, "eval_sampling/importance_sampling_ratio/mean": 1.0161973595619203, "eval_sampling/importance_sampling_ratio/max": 1.4135438442230224, "eval_entropy": 0.6903003931045533, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7133846759796143, "eval_reward_meter_mean": 0.8406145215034485, "eval_reward_meter_std": 0.21775756597053259, "eval_reward_count_adherence_mean": 0.9806249916553498, "eval_reward_count_adherence_std": 0.049967176467180255, "eval_reward_hard_gate_mean": 0.9625, "eval_reward_hard_gate_std": 0.10606601536273956, "eval_reward_repeat_soft_mean": 0.9468575239181518, "eval_reward_repeat_soft_std": 0.05658040288835764, "eval_reward_judge_quality_mean": 0.36937499046325684, "eval_reward_judge_quality_std": 0.1088132050819695, "eval_reward_total_composite_mean": 0.7133846759796143, "eval_reward_total_composite_std": 0.1489717047661543} {"timestamp_utc": "2026-04-13T04:06:44Z", "mode": "train", "global_step": 2301, "epoch": 0.2311401305876444, "loss": 0.09, "grad_norm": 16.017528533935547, "learning_rate": 3.0303030303030305e-06, "num_tokens": 4241588.0, "completions/mean_length": 33.125, "completions/min_length": 29.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.125, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.749167799949646, "rewards/meter/std": 0.3175085186958313, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9394254684448242, "rewards/repeat_soft/std": 0.040412817150354385, "rewards/judge_quality/mean": 0.9200000166893005, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8570680618286133, "rewards/total_composite/std": 0.14137758314609528, "reward": 0.8570680618286133, "reward_std": 0.14137756824493408, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1226234883069992, "sampling/sampling_logp_difference/max": 1.834824800491333, "sampling/importance_sampling_ratio/min": 0.15964147448539734, "sampling/importance_sampling_ratio/mean": 0.9949538111686707, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7573369964957237, "clip_ratio/low_mean": 0.017608359223231673, "clip_ratio/low_min": 0.017608359223231673, "clip_ratio/high_mean": 0.0785303795710206, "clip_ratio/high_max": 0.0785303795710206, "clip_ratio/region_mean": 0.09613873879425228, "reward_total_mean": 0.8570680618286133, "reward_meter_mean": 0.749167799949646, "reward_meter_std": 0.3175085186958313, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9394254684448242, "reward_repeat_soft_std": 0.040412817150354385, "reward_judge_quality_mean": 0.9200000166893005, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8570680618286133, "reward_total_composite_std": 0.14137758314609528} {"timestamp_utc": "2026-04-13T04:06:51Z", "mode": "train", "global_step": 2302, "epoch": 0.23124058262179809, "loss": 0.048, "grad_norm": 6.956501483917236, "learning_rate": 3.0272727272727277e-06, "num_tokens": 4243748.0, "completions/mean_length": 108.0, "completions/min_length": 78.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.0, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.9211714267730713, "rewards/meter/std": 0.034196846187114716, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9528183937072754, "rewards/repeat_soft/std": 0.036325905472040176, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7402464747428894, "rewards/total_composite/std": 0.027572114020586014, "reward": 0.7402464747428894, "reward_std": 0.027572132647037506, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09399783611297607, "sampling/sampling_logp_difference/max": 1.6350078582763672, "sampling/importance_sampling_ratio/min": 0.19495083391666412, "sampling/importance_sampling_ratio/mean": 1.0152182579040527, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5805242471396923, "clip_ratio/low_mean": 0.024357769638299942, "clip_ratio/low_min": 0.024357769638299942, "clip_ratio/high_mean": 0.0576205572579056, "clip_ratio/high_max": 0.0576205572579056, "clip_ratio/region_mean": 0.08197832689620554, "reward_total_mean": 0.7402464747428894, "reward_meter_mean": 0.9211714267730713, "reward_meter_std": 0.034196846187114716, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9528183937072754, "reward_repeat_soft_std": 0.036325905472040176, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7402464747428894, "reward_total_composite_std": 0.027572114020586014} {"timestamp_utc": "2026-04-13T04:06:58Z", "mode": "train", "global_step": 2303, "epoch": 0.2313410346559518, "loss": -0.0152, "grad_norm": 10.340105056762695, "learning_rate": 3.0242424242424246e-06, "num_tokens": 4245440.0, "completions/mean_length": 66.5, "completions/min_length": 62.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9851570129394531, "rewards/meter/std": 0.013062767684459686, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9854460954666138, "rewards/repeat_soft/std": 0.016635842621326447, "rewards/judge_quality/mean": 0.5049999952316284, "rewards/judge_quality/std": 0.16801361739635468, "rewards/total_composite/mean": 0.8433653116226196, "rewards/total_composite/std": 0.046466659754514694, "reward": 0.8433653116226196, "reward_std": 0.0464666523039341, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12231482565402985, "sampling/sampling_logp_difference/max": 1.0127854347229004, "sampling/importance_sampling_ratio/min": 0.3632059097290039, "sampling/importance_sampling_ratio/mean": 1.0181692838668823, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0306510403752327, "clip_ratio/low_mean": 0.07971266750246286, "clip_ratio/low_min": 0.07971266750246286, "clip_ratio/high_mean": 0.017605634406208992, "clip_ratio/high_max": 0.017605634406208992, "clip_ratio/region_mean": 0.09731830190867186, "reward_total_mean": 0.8433653116226196, "reward_meter_mean": 0.9851570129394531, "reward_meter_std": 0.013062767684459686, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9854460954666138, "reward_repeat_soft_std": 0.016635842621326447, "reward_judge_quality_mean": 0.5049999952316284, "reward_judge_quality_std": 0.16801361739635468, "reward_total_composite_mean": 0.8433653116226196, "reward_total_composite_std": 0.046466659754514694} {"timestamp_utc": "2026-04-13T04:07:10Z", "mode": "train", "global_step": 2304, "epoch": 0.23144148669010547, "loss": -0.1271, "grad_norm": 2.53794264793396, "learning_rate": 3.0212121212121214e-06, "num_tokens": 4247130.0, "completions/mean_length": 113.25, "completions/min_length": 51.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 56.28571701049805, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.7727634906768799, "rewards/meter/std": 0.39990365505218506, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.987604022026062, "rewards/repeat_soft/std": 0.013574519194662571, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.13617216050624847, "rewards/total_composite/mean": 0.677879273891449, "rewards/total_composite/std": 0.298076331615448, "reward": 0.677879273891449, "reward_std": 0.2980763018131256, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13464616239070892, "sampling/sampling_logp_difference/max": 2.019390344619751, "sampling/importance_sampling_ratio/min": 0.13273636996746063, "sampling/importance_sampling_ratio/mean": 1.0187554359436035, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7828288674354553, "clip_ratio/low_mean": 0.008064515888690948, "clip_ratio/low_min": 0.008064515888690948, "clip_ratio/high_mean": 0.09406738076359034, "clip_ratio/high_max": 0.09406738076359034, "clip_ratio/region_mean": 0.10213189665228128, "reward_total_mean": 0.677879273891449, "reward_meter_mean": 0.7727634906768799, "reward_meter_std": 0.39990365505218506, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.987604022026062, "reward_repeat_soft_std": 0.013574519194662571, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.13617216050624847, "reward_total_composite_mean": 0.677879273891449, "reward_total_composite_std": 0.298076331615448} {"timestamp_utc": "2026-04-13T04:07:16Z", "mode": "train", "global_step": 2305, "epoch": 0.23154193872425916, "loss": 0.076, "grad_norm": 24.02999496459961, "learning_rate": 3.0181818181818182e-06, "num_tokens": 4248640.0, "completions/mean_length": 22.75, "completions/min_length": 18.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.75, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.9085512757301331, "rewards/meter/std": 0.12152045965194702, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9611742496490479, "rewards/repeat_soft/std": 0.0037498052697628736, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.8559654951095581, "rewards/total_composite/std": 0.07772092521190643, "reward": 0.8559654951095581, "reward_std": 0.07772091776132584, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14636266231536865, "sampling/sampling_logp_difference/max": 1.530447006225586, "sampling/importance_sampling_ratio/min": 0.2164389044046402, "sampling/importance_sampling_ratio/mean": 1.0124561786651611, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7292656749486923, "clip_ratio/low_mean": 0.045129282865673304, "clip_ratio/low_min": 0.045129282865673304, "clip_ratio/high_mean": 0.035774410236626863, "clip_ratio/high_max": 0.035774410236626863, "clip_ratio/region_mean": 0.08090369310230017, "reward_total_mean": 0.8559654951095581, "reward_meter_mean": 0.9085512757301331, "reward_meter_std": 0.12152045965194702, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9611742496490479, "reward_repeat_soft_std": 0.0037498052697628736, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.8559654951095581, "reward_total_composite_std": 0.07772092521190643} {"timestamp_utc": "2026-04-13T04:07:23Z", "mode": "train", "global_step": 2306, "epoch": 0.23164239075841286, "loss": 0.0331, "grad_norm": 10.666251182556152, "learning_rate": 3.0151515151515155e-06, "num_tokens": 4250372.0, "completions/mean_length": 65.5, "completions/min_length": 55.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.5, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.5885916948318481, "rewards/meter/std": 0.32950305938720703, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9884371757507324, "rewards/repeat_soft/std": 0.015632735565304756, "rewards/judge_quality/mean": 0.45125001668930054, "rewards/judge_quality/std": 0.12799973785877228, "rewards/total_composite/mean": 0.6428350210189819, "rewards/total_composite/std": 0.16123047471046448, "reward": 0.6428350210189819, "reward_std": 0.16123047471046448, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14198045432567596, "sampling/sampling_logp_difference/max": 1.493424654006958, "sampling/importance_sampling_ratio/min": 0.22460214793682098, "sampling/importance_sampling_ratio/mean": 1.018308162689209, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.812913566827774, "clip_ratio/low_mean": 0.07348484918475151, "clip_ratio/low_min": 0.07348484918475151, "clip_ratio/high_mean": 0.06532427482306957, "clip_ratio/high_max": 0.06532427482306957, "clip_ratio/region_mean": 0.13880912400782108, "reward_total_mean": 0.6428350210189819, "reward_meter_mean": 0.5885916948318481, "reward_meter_std": 0.32950305938720703, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9884371757507324, "reward_repeat_soft_std": 0.015632735565304756, "reward_judge_quality_mean": 0.45125001668930054, "reward_judge_quality_std": 0.12799973785877228, "reward_total_composite_mean": 0.6428350210189819, "reward_total_composite_std": 0.16123047471046448} {"timestamp_utc": "2026-04-13T04:07:34Z", "mode": "train", "global_step": 2307, "epoch": 0.23174284279256654, "loss": -0.1333, "grad_norm": 2.8561487197875977, "learning_rate": 3.0121212121212123e-06, "num_tokens": 4252159.0, "completions/mean_length": 126.375, "completions/min_length": 62.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 71.28572082519531, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.7502450346946716, "rewards/meter/std": 0.30973419547080994, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9829877018928528, "rewards/repeat_soft/std": 0.011588296853005886, "rewards/judge_quality/mean": 0.3399999737739563, "rewards/judge_quality/std": 0.15052290260791779, "rewards/total_composite/mean": 0.6001917123794556, "rewards/total_composite/std": 0.2854815125465393, "reward": 0.6001917123794556, "reward_std": 0.2854815125465393, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1099759042263031, "sampling/sampling_logp_difference/max": 1.177825927734375, "sampling/importance_sampling_ratio/min": 0.3079475164413452, "sampling/importance_sampling_ratio/mean": 1.014804482460022, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.632517009973526, "clip_ratio/low_mean": 0.04438405856490135, "clip_ratio/low_min": 0.04438405856490135, "clip_ratio/high_mean": 0.08052795380353928, "clip_ratio/high_max": 0.08052795380353928, "clip_ratio/region_mean": 0.12491201236844063, "reward_total_mean": 0.6001917123794556, "reward_meter_mean": 0.7502450346946716, "reward_meter_std": 0.30973419547080994, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9829877018928528, "reward_repeat_soft_std": 0.011588296853005886, "reward_judge_quality_mean": 0.3399999737739563, "reward_judge_quality_std": 0.15052290260791779, "reward_total_composite_mean": 0.6001917123794556, "reward_total_composite_std": 0.2854815125465393} {"timestamp_utc": "2026-04-13T04:07:41Z", "mode": "train", "global_step": 2308, "epoch": 0.23184329482672025, "loss": 0.0007, "grad_norm": 6.851720809936523, "learning_rate": 3.009090909090909e-06, "num_tokens": 4254050.0, "completions/mean_length": 72.375, "completions/min_length": 66.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.375, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9880704879760742, "rewards/meter/std": 0.005852893926203251, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9680160284042358, "rewards/repeat_soft/std": 0.02902626246213913, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.1940544992685318, "rewards/total_composite/mean": 0.8309333324432373, "rewards/total_composite/std": 0.05927732586860657, "reward": 0.8309333324432373, "reward_std": 0.059277310967445374, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10497382283210754, "sampling/sampling_logp_difference/max": 1.2197494506835938, "sampling/importance_sampling_ratio/min": 0.29530414938926697, "sampling/importance_sampling_ratio/mean": 1.0043128728866577, "sampling/importance_sampling_ratio/max": 1.8544739484786987, "entropy": 0.6177698150277138, "clip_ratio/low_mean": 0.07934173848479986, "clip_ratio/low_min": 0.07934173848479986, "clip_ratio/high_mean": 0.00844594556838274, "clip_ratio/high_max": 0.00844594556838274, "clip_ratio/region_mean": 0.0877876840531826, "reward_total_mean": 0.8309333324432373, "reward_meter_mean": 0.9880704879760742, "reward_meter_std": 0.005852893926203251, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9680160284042358, "reward_repeat_soft_std": 0.02902626246213913, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.1940544992685318, "reward_total_composite_mean": 0.8309333324432373, "reward_total_composite_std": 0.05927732586860657} {"timestamp_utc": "2026-04-13T04:07:48Z", "mode": "train", "global_step": 2309, "epoch": 0.23194374686087393, "loss": -0.0367, "grad_norm": 7.349184513092041, "learning_rate": 3.0060606060606064e-06, "num_tokens": 4255800.0, "completions/mean_length": 61.75, "completions/min_length": 54.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.75, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.8977428674697876, "rewards/meter/std": 0.212468221783638, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9693663716316223, "rewards/repeat_soft/std": 0.028105057775974274, "rewards/judge_quality/mean": 0.42624998092651367, "rewards/judge_quality/std": 0.21980105340480804, "rewards/total_composite/mean": 0.7787959575653076, "rewards/total_composite/std": 0.12500521540641785, "reward": 0.7787959575653076, "reward_std": 0.12500521540641785, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11913736909627914, "sampling/sampling_logp_difference/max": 1.6229333877563477, "sampling/importance_sampling_ratio/min": 0.19731903076171875, "sampling/importance_sampling_ratio/mean": 1.0028271675109863, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7619753405451775, "clip_ratio/low_mean": 0.036897131241858006, "clip_ratio/low_min": 0.036897131241858006, "clip_ratio/high_mean": 0.09098266437649727, "clip_ratio/high_max": 0.09098266437649727, "clip_ratio/region_mean": 0.12787979561835527, "reward_total_mean": 0.7787959575653076, "reward_meter_mean": 0.8977428674697876, "reward_meter_std": 0.212468221783638, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9693663716316223, "reward_repeat_soft_std": 0.028105057775974274, "reward_judge_quality_mean": 0.42624998092651367, "reward_judge_quality_std": 0.21980105340480804, "reward_total_composite_mean": 0.7787959575653076, "reward_total_composite_std": 0.12500521540641785} {"timestamp_utc": "2026-04-13T04:07:55Z", "mode": "train", "global_step": 2310, "epoch": 0.23204419889502761, "loss": 0.0011, "grad_norm": 10.889214515686035, "learning_rate": 3.0030303030303032e-06, "num_tokens": 4257239.0, "completions/mean_length": 31.875, "completions/min_length": 26.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.875, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9208740592002869, "rewards/meter/std": 0.17093372344970703, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9336026906967163, "rewards/repeat_soft/std": 0.045483969151973724, "rewards/judge_quality/mean": 0.42499998211860657, "rewards/judge_quality/std": 0.0707106739282608, "rewards/total_composite/mean": 0.785253643989563, "rewards/total_composite/std": 0.07501427829265594, "reward": 0.785253643989563, "reward_std": 0.07501427084207535, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09354891628026962, "sampling/sampling_logp_difference/max": 0.9078412055969238, "sampling/importance_sampling_ratio/min": 0.4162488281726837, "sampling/importance_sampling_ratio/mean": 1.0384480953216553, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6584621965885162, "clip_ratio/low_mean": 0.04095982201397419, "clip_ratio/low_min": 0.04095982201397419, "clip_ratio/high_mean": 0.08022548258304596, "clip_ratio/high_max": 0.08022548258304596, "clip_ratio/region_mean": 0.12118530459702015, "reward_total_mean": 0.785253643989563, "reward_meter_mean": 0.9208740592002869, "reward_meter_std": 0.17093372344970703, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9336026906967163, "reward_repeat_soft_std": 0.045483969151973724, "reward_judge_quality_mean": 0.42499998211860657, "reward_judge_quality_std": 0.0707106739282608, "reward_total_composite_mean": 0.785253643989563, "reward_total_composite_std": 0.07501427829265594} {"timestamp_utc": "2026-04-13T04:08:07Z", "mode": "train", "global_step": 2311, "epoch": 0.23214465092918132, "loss": -0.1534, "grad_norm": 2.0654704570770264, "learning_rate": 3e-06, "num_tokens": 4259049.0, "completions/mean_length": 123.25, "completions/min_length": 57.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 67.71428680419922, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.7906265258789062, "rewards/meter/std": 0.28741228580474854, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9778039455413818, "rewards/repeat_soft/std": 0.018559660762548447, "rewards/judge_quality/mean": 0.3187499940395355, "rewards/judge_quality/std": 0.14961257576942444, "rewards/total_composite/mean": 0.6576589345932007, "rewards/total_composite/std": 0.2760924994945526, "reward": 0.6576589345932007, "reward_std": 0.2760924994945526, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10000292211771011, "sampling/sampling_logp_difference/max": 0.9483463764190674, "sampling/importance_sampling_ratio/min": 0.38738107681274414, "sampling/importance_sampling_ratio/mean": 1.0073597431182861, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5746269896626472, "clip_ratio/low_mean": 0.0016891892300918698, "clip_ratio/low_min": 0.0016891892300918698, "clip_ratio/high_mean": 0.08898234367370605, "clip_ratio/high_max": 0.08898234367370605, "clip_ratio/region_mean": 0.09067153290379792, "reward_total_mean": 0.6576589345932007, "reward_meter_mean": 0.7906265258789062, "reward_meter_std": 0.28741228580474854, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9778039455413818, "reward_repeat_soft_std": 0.018559660762548447, "reward_judge_quality_mean": 0.3187499940395355, "reward_judge_quality_std": 0.14961257576942444, "reward_total_composite_mean": 0.6576589345932007, "reward_total_composite_std": 0.2760924994945526} {"timestamp_utc": "2026-04-13T04:08:17Z", "mode": "train", "global_step": 2312, "epoch": 0.232245102963335, "loss": 0.0273, "grad_norm": 11.839028358459473, "learning_rate": 2.996969696969697e-06, "num_tokens": 4260834.0, "completions/mean_length": 59.125, "completions/min_length": 54.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.125, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9495798349380493, "rewards/meter/std": 0.08213601261377335, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9681565761566162, "rewards/repeat_soft/std": 0.046142950654029846, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.08166787773370743, "rewards/total_composite/mean": 0.7885016202926636, "rewards/total_composite/std": 0.059757739305496216, "reward": 0.7885016202926636, "reward_std": 0.05975772812962532, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10804279148578644, "sampling/sampling_logp_difference/max": 1.4481281042099, "sampling/importance_sampling_ratio/min": 0.2350097894668579, "sampling/importance_sampling_ratio/mean": 0.9923175573348999, "sampling/importance_sampling_ratio/max": 1.8408204317092896, "entropy": 0.7042853161692619, "clip_ratio/low_mean": 0.027236653491854668, "clip_ratio/low_min": 0.027236653491854668, "clip_ratio/high_mean": 0.10615945886820555, "clip_ratio/high_max": 0.10615945886820555, "clip_ratio/region_mean": 0.13339611236006021, "reward_total_mean": 0.7885016202926636, "reward_meter_mean": 0.9495798349380493, "reward_meter_std": 0.08213601261377335, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9681565761566162, "reward_repeat_soft_std": 0.046142950654029846, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.08166787773370743, "reward_total_composite_mean": 0.7885016202926636, "reward_total_composite_std": 0.059757739305496216} {"timestamp_utc": "2026-04-13T04:08:28Z", "mode": "train", "global_step": 2313, "epoch": 0.2323455549974887, "loss": 0.0356, "grad_norm": 4.7070817947387695, "learning_rate": 2.993939393939394e-06, "num_tokens": 4263595.0, "completions/mean_length": 165.125, "completions/min_length": 148.0, "completions/max_length": 190.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 165.125, "completions/min_terminated_length": 148.0, "completions/max_terminated_length": 190.0, "rewards/meter/mean": 0.9741771221160889, "rewards/meter/std": 0.009559054858982563, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7797747850418091, "rewards/repeat_soft/std": 0.06179782748222351, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.763107180595398, "rewards/total_composite/std": 0.03719240799546242, "reward": 0.763107180595398, "reward_std": 0.03719239681959152, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08506263047456741, "sampling/sampling_logp_difference/max": 2.8313820362091064, "sampling/importance_sampling_ratio/min": 0.05893135443329811, "sampling/importance_sampling_ratio/mean": 1.0103092193603516, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47486183419823647, "clip_ratio/low_mean": 0.03735977178439498, "clip_ratio/low_min": 0.03735977178439498, "clip_ratio/high_mean": 0.03549026045948267, "clip_ratio/high_max": 0.03549026045948267, "clip_ratio/region_mean": 0.07285003224387765, "reward_total_mean": 0.763107180595398, "reward_meter_mean": 0.9741771221160889, "reward_meter_std": 0.009559054858982563, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7797747850418091, "reward_repeat_soft_std": 0.06179782748222351, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.763107180595398, "reward_total_composite_std": 0.03719240799546242} {"timestamp_utc": "2026-04-13T04:08:44Z", "mode": "train", "global_step": 2314, "epoch": 0.2324460070316424, "loss": -0.1544, "grad_norm": 1.7457951307296753, "learning_rate": 2.990909090909091e-06, "num_tokens": 4265416.0, "completions/mean_length": 118.625, "completions/min_length": 51.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 62.42857360839844, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.85625159740448, "rewards/meter/std": 0.33597326278686523, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9877939224243164, "rewards/repeat_soft/std": 0.010522305965423584, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.15315140783786774, "rewards/total_composite/mean": 0.7026214599609375, "rewards/total_composite/std": 0.28649455308914185, "reward": 0.7026214599609375, "reward_std": 0.28649455308914185, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13589252531528473, "sampling/sampling_logp_difference/max": 1.3575525283813477, "sampling/importance_sampling_ratio/min": 0.2589673697948456, "sampling/importance_sampling_ratio/mean": 1.0150023698806763, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6534537747502327, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09480732260271907, "clip_ratio/high_max": 0.09480732260271907, "clip_ratio/region_mean": 0.09480732260271907, "reward_total_mean": 0.7026214599609375, "reward_meter_mean": 0.85625159740448, "reward_meter_std": 0.33597326278686523, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9877939224243164, "reward_repeat_soft_std": 0.010522305965423584, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.15315140783786774, "reward_total_composite_mean": 0.7026214599609375, "reward_total_composite_std": 0.28649455308914185} {"timestamp_utc": "2026-04-13T04:08:52Z", "mode": "train", "global_step": 2315, "epoch": 0.23254645906579607, "loss": -0.0162, "grad_norm": 5.999134540557861, "learning_rate": 2.987878787878788e-06, "num_tokens": 4268019.0, "completions/mean_length": 150.375, "completions/min_length": 126.0, "completions/max_length": 169.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 150.375, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 169.0, "rewards/meter/mean": 0.9846161007881165, "rewards/meter/std": 0.013704297132790089, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.923574686050415, "rewards/repeat_soft/std": 0.03638064116239548, "rewards/judge_quality/mean": 0.2462499886751175, "rewards/judge_quality/std": 0.08348438143730164, "rewards/total_composite/mean": 0.7555597424507141, "rewards/total_composite/std": 0.026301415637135506, "reward": 0.7555597424507141, "reward_std": 0.026301424950361252, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10009395331144333, "sampling/sampling_logp_difference/max": 1.4756245613098145, "sampling/importance_sampling_ratio/min": 0.22863589227199554, "sampling/importance_sampling_ratio/mean": 1.0066783428192139, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6125488467514515, "clip_ratio/low_mean": 0.032616401091217995, "clip_ratio/low_min": 0.032616401091217995, "clip_ratio/high_mean": 0.0727042518556118, "clip_ratio/high_max": 0.0727042518556118, "clip_ratio/region_mean": 0.1053206529468298, "reward_total_mean": 0.7555597424507141, "reward_meter_mean": 0.9846161007881165, "reward_meter_std": 0.013704297132790089, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.923574686050415, "reward_repeat_soft_std": 0.03638064116239548, "reward_judge_quality_mean": 0.2462499886751175, "reward_judge_quality_std": 0.08348438143730164, "reward_total_composite_mean": 0.7555597424507141, "reward_total_composite_std": 0.026301415637135506} {"timestamp_utc": "2026-04-13T04:09:00Z", "mode": "train", "global_step": 2316, "epoch": 0.23264691109994978, "loss": 0.0448, "grad_norm": 5.673974990844727, "learning_rate": 2.984848484848485e-06, "num_tokens": 4270727.0, "completions/mean_length": 144.5, "completions/min_length": 129.0, "completions/max_length": 155.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 144.5, "completions/min_terminated_length": 129.0, "completions/max_terminated_length": 155.0, "rewards/meter/mean": 0.986276388168335, "rewards/meter/std": 0.003829032415524125, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8583556413650513, "rewards/repeat_soft/std": 0.03872491419315338, "rewards/judge_quality/mean": 0.3137499690055847, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.773784875869751, "rewards/total_composite/std": 0.02564755454659462, "reward": 0.773784875869751, "reward_std": 0.02564755268394947, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10662887990474701, "sampling/sampling_logp_difference/max": 2.1683647632598877, "sampling/importance_sampling_ratio/min": 0.11436448246240616, "sampling/importance_sampling_ratio/mean": 1.010918140411377, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5708662942051888, "clip_ratio/low_mean": 0.06946881022304296, "clip_ratio/low_min": 0.06946881022304296, "clip_ratio/high_mean": 0.04276931285858154, "clip_ratio/high_max": 0.04276931285858154, "clip_ratio/region_mean": 0.11223812308162451, "reward_total_mean": 0.773784875869751, "reward_meter_mean": 0.986276388168335, "reward_meter_std": 0.003829032415524125, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8583556413650513, "reward_repeat_soft_std": 0.03872491419315338, "reward_judge_quality_mean": 0.3137499690055847, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.773784875869751, "reward_total_composite_std": 0.02564755454659462} {"timestamp_utc": "2026-04-13T04:09:06Z", "mode": "train", "global_step": 2317, "epoch": 0.23274736313410346, "loss": 0.0268, "grad_norm": 12.251059532165527, "learning_rate": 2.981818181818182e-06, "num_tokens": 4272318.0, "completions/mean_length": 43.875, "completions/min_length": 41.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.875, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.8647700548171997, "rewards/meter/std": 0.2843247652053833, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9859254360198975, "rewards/repeat_soft/std": 0.014082822017371655, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.7869890332221985, "rewards/total_composite/std": 0.14336487650871277, "reward": 0.7869890332221985, "reward_std": 0.14336487650871277, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1057528406381607, "sampling/sampling_logp_difference/max": 1.2945847511291504, "sampling/importance_sampling_ratio/min": 0.27401161193847656, "sampling/importance_sampling_ratio/mean": 1.0113645792007446, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5018568336963654, "clip_ratio/low_mean": 0.008522727526724339, "clip_ratio/low_min": 0.008522727526724339, "clip_ratio/high_mean": 0.07204574206843972, "clip_ratio/high_max": 0.07204574206843972, "clip_ratio/region_mean": 0.08056846959516406, "reward_total_mean": 0.7869890332221985, "reward_meter_mean": 0.8647700548171997, "reward_meter_std": 0.2843247652053833, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9859254360198975, "reward_repeat_soft_std": 0.014082822017371655, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.7869890332221985, "reward_total_composite_std": 0.14336487650871277} {"timestamp_utc": "2026-04-13T04:09:13Z", "mode": "train", "global_step": 2318, "epoch": 0.23284781516825714, "loss": 0.0421, "grad_norm": 9.483677864074707, "learning_rate": 2.9787878787878787e-06, "num_tokens": 4274269.0, "completions/mean_length": 82.875, "completions/min_length": 76.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.875, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.9470477104187012, "rewards/meter/std": 0.0720118060708046, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9662922024726868, "rewards/repeat_soft/std": 0.02239467203617096, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7988007068634033, "rewards/total_composite/std": 0.032803282141685486, "reward": 0.7988007068634033, "reward_std": 0.03280327469110489, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12078768014907837, "sampling/sampling_logp_difference/max": 1.2818994522094727, "sampling/importance_sampling_ratio/min": 0.2775096893310547, "sampling/importance_sampling_ratio/mean": 1.014748215675354, "sampling/importance_sampling_ratio/max": 1.9293464422225952, "entropy": 0.7252351641654968, "clip_ratio/low_mean": 0.03045515436679125, "clip_ratio/low_min": 0.03045515436679125, "clip_ratio/high_mean": 0.08145811222493649, "clip_ratio/high_max": 0.08145811222493649, "clip_ratio/region_mean": 0.11191326659172773, "reward_total_mean": 0.7988007068634033, "reward_meter_mean": 0.9470477104187012, "reward_meter_std": 0.0720118060708046, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9662922024726868, "reward_repeat_soft_std": 0.02239467203617096, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7988007068634033, "reward_total_composite_std": 0.032803282141685486} {"timestamp_utc": "2026-04-13T04:09:20Z", "mode": "train", "global_step": 2319, "epoch": 0.23294826720241085, "loss": 0.0268, "grad_norm": 11.107125282287598, "learning_rate": 2.9757575757575756e-06, "num_tokens": 4276245.0, "completions/mean_length": 62.0, "completions/min_length": 55.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9892219305038452, "rewards/meter/std": 0.01330061350017786, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9567365646362305, "rewards/repeat_soft/std": 0.023429470136761665, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.0843462198972702, "rewards/total_composite/mean": 0.8063235282897949, "rewards/total_composite/std": 0.029908373951911926, "reward": 0.8063235282897949, "reward_std": 0.029908375814557076, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10060884058475494, "sampling/sampling_logp_difference/max": 1.340210199356079, "sampling/importance_sampling_ratio/min": 0.2617906332015991, "sampling/importance_sampling_ratio/mean": 1.0096288919448853, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7003774605691433, "clip_ratio/low_mean": 0.01707535912282765, "clip_ratio/low_min": 0.01707535912282765, "clip_ratio/high_mean": 0.06571874395012856, "clip_ratio/high_max": 0.06571874395012856, "clip_ratio/region_mean": 0.0827941030729562, "reward_total_mean": 0.8063235282897949, "reward_meter_mean": 0.9892219305038452, "reward_meter_std": 0.01330061350017786, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9567365646362305, "reward_repeat_soft_std": 0.023429470136761665, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.0843462198972702, "reward_total_composite_mean": 0.8063235282897949, "reward_total_composite_std": 0.029908373951911926} {"timestamp_utc": "2026-04-13T04:09:26Z", "mode": "train", "global_step": 2320, "epoch": 0.23304871923656453, "loss": 0.04, "grad_norm": 14.92398452758789, "learning_rate": 2.9727272727272733e-06, "num_tokens": 4277832.0, "completions/mean_length": 40.375, "completions/min_length": 35.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.375, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.6923952102661133, "rewards/meter/std": 0.375019371509552, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9732975959777832, "rewards/repeat_soft/std": 0.033407121896743774, "rewards/judge_quality/mean": 0.5900000333786011, "rewards/judge_quality/std": 0.27994900941848755, "rewards/total_composite/mean": 0.7359076142311096, "rewards/total_composite/std": 0.17495134472846985, "reward": 0.7359076142311096, "reward_std": 0.17495134472846985, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1265413910150528, "sampling/sampling_logp_difference/max": 2.266915798187256, "sampling/importance_sampling_ratio/min": 0.10363131016492844, "sampling/importance_sampling_ratio/mean": 1.0151102542877197, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5555492043495178, "clip_ratio/low_mean": 0.05653615854680538, "clip_ratio/low_min": 0.05653615854680538, "clip_ratio/high_mean": 0.028252556221559644, "clip_ratio/high_max": 0.028252556221559644, "clip_ratio/region_mean": 0.08478871476836503, "reward_total_mean": 0.7359076142311096, "reward_meter_mean": 0.6923952102661133, "reward_meter_std": 0.375019371509552, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9732975959777832, "reward_repeat_soft_std": 0.033407121896743774, "reward_judge_quality_mean": 0.5900000333786011, "reward_judge_quality_std": 0.27994900941848755, "reward_total_composite_mean": 0.7359076142311096, "reward_total_composite_std": 0.17495134472846985} {"timestamp_utc": "2026-04-13T04:09:38Z", "mode": "train", "global_step": 2321, "epoch": 0.23314917127071824, "loss": -0.1306, "grad_norm": 3.446657419204712, "learning_rate": 2.96969696969697e-06, "num_tokens": 4279758.0, "completions/mean_length": 129.75, "completions/min_length": 66.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 75.14286041259766, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.7992166876792908, "rewards/meter/std": 0.3218916356563568, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8914412260055542, "rewards/repeat_soft/std": 0.08550725877285004, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.6175141334533691, "rewards/total_composite/std": 0.2912995219230652, "reward": 0.6175141334533691, "reward_std": 0.2912995219230652, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1072724238038063, "sampling/sampling_logp_difference/max": 1.8615368604660034, "sampling/importance_sampling_ratio/min": 0.1554335653781891, "sampling/importance_sampling_ratio/mean": 0.9897790551185608, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36906541883945465, "clip_ratio/low_mean": 0.018563789315521717, "clip_ratio/low_min": 0.018563789315521717, "clip_ratio/high_mean": 0.05659684259444475, "clip_ratio/high_max": 0.05659684259444475, "clip_ratio/region_mean": 0.07516063190996647, "reward_total_mean": 0.6175141334533691, "reward_meter_mean": 0.7992166876792908, "reward_meter_std": 0.3218916356563568, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8914412260055542, "reward_repeat_soft_std": 0.08550725877285004, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.6175141334533691, "reward_total_composite_std": 0.2912995219230652} {"timestamp_utc": "2026-04-13T04:09:46Z", "mode": "train", "global_step": 2322, "epoch": 0.23324962330487192, "loss": 0.0146, "grad_norm": 5.813755035400391, "learning_rate": 2.9666666666666673e-06, "num_tokens": 4282395.0, "completions/mean_length": 149.625, "completions/min_length": 139.0, "completions/max_length": 161.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 149.625, "completions/min_terminated_length": 139.0, "completions/max_terminated_length": 161.0, "rewards/meter/mean": 0.9172002077102661, "rewards/meter/std": 0.1912015974521637, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9265111684799194, "rewards/repeat_soft/std": 0.03402803838253021, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.7648912072181702, "rewards/total_composite/std": 0.08437686413526535, "reward": 0.7648912072181702, "reward_std": 0.08437685668468475, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09510429948568344, "sampling/sampling_logp_difference/max": 1.8696269989013672, "sampling/importance_sampling_ratio/min": 0.15418116748332977, "sampling/importance_sampling_ratio/mean": 1.0155636072158813, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6255929693579674, "clip_ratio/low_mean": 0.028946529142558575, "clip_ratio/low_min": 0.028946529142558575, "clip_ratio/high_mean": 0.05362664069980383, "clip_ratio/high_max": 0.05362664069980383, "clip_ratio/region_mean": 0.0825731698423624, "reward_total_mean": 0.7648912072181702, "reward_meter_mean": 0.9172002077102661, "reward_meter_std": 0.1912015974521637, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9265111684799194, "reward_repeat_soft_std": 0.03402803838253021, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.7648912072181702, "reward_total_composite_std": 0.08437686413526535} {"timestamp_utc": "2026-04-13T04:09:52Z", "mode": "train", "global_step": 2323, "epoch": 0.2333500753390256, "loss": 0.021, "grad_norm": 10.274893760681152, "learning_rate": 2.963636363636364e-06, "num_tokens": 4284266.0, "completions/mean_length": 69.875, "completions/min_length": 67.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.875, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.6226397752761841, "rewards/meter/std": 0.3947320580482483, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9777458310127258, "rewards/repeat_soft/std": 0.028268402442336082, "rewards/judge_quality/mean": 0.5649999976158142, "rewards/judge_quality/std": 0.194054514169693, "rewards/total_composite/mean": 0.6974624395370483, "rewards/total_composite/std": 0.16974987089633942, "reward": 0.6974624395370483, "reward_std": 0.16974985599517822, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11650034785270691, "sampling/sampling_logp_difference/max": 1.8720250129699707, "sampling/importance_sampling_ratio/min": 0.15381188690662384, "sampling/importance_sampling_ratio/mean": 1.0221940279006958, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7056416794657707, "clip_ratio/low_mean": 0.05850917100906372, "clip_ratio/low_min": 0.05850917100906372, "clip_ratio/high_mean": 0.046669671311974525, "clip_ratio/high_max": 0.046669671311974525, "clip_ratio/region_mean": 0.10517884232103825, "reward_total_mean": 0.6974624395370483, "reward_meter_mean": 0.6226397752761841, "reward_meter_std": 0.3947320580482483, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9777458310127258, "reward_repeat_soft_std": 0.028268402442336082, "reward_judge_quality_mean": 0.5649999976158142, "reward_judge_quality_std": 0.194054514169693, "reward_total_composite_mean": 0.6974624395370483, "reward_total_composite_std": 0.16974987089633942} {"timestamp_utc": "2026-04-13T04:09:59Z", "mode": "train", "global_step": 2324, "epoch": 0.2334505273731793, "loss": 0.0013, "grad_norm": 6.046849250793457, "learning_rate": 2.960606060606061e-06, "num_tokens": 4286507.0, "completions/mean_length": 102.125, "completions/min_length": 91.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.125, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.9873858690261841, "rewards/meter/std": 0.0025377804413437843, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8010890483856201, "rewards/repeat_soft/std": 0.03700441122055054, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7876825332641602, "rewards/total_composite/std": 0.026109006255865097, "reward": 0.7876825332641602, "reward_std": 0.0261089988052845, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08201354742050171, "sampling/sampling_logp_difference/max": 2.3267884254455566, "sampling/importance_sampling_ratio/min": 0.09760871529579163, "sampling/importance_sampling_ratio/mean": 1.000362515449524, "sampling/importance_sampling_ratio/max": 1.9933643341064453, "entropy": 0.41017764061689377, "clip_ratio/low_mean": 0.019166581332683563, "clip_ratio/low_min": 0.019166581332683563, "clip_ratio/high_mean": 0.06503412453457713, "clip_ratio/high_max": 0.06503412453457713, "clip_ratio/region_mean": 0.0842007058672607, "reward_total_mean": 0.7876825332641602, "reward_meter_mean": 0.9873858690261841, "reward_meter_std": 0.0025377804413437843, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8010890483856201, "reward_repeat_soft_std": 0.03700441122055054, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7876825332641602, "reward_total_composite_std": 0.026109006255865097} {"timestamp_utc": "2026-04-13T04:10:07Z", "mode": "train", "global_step": 2325, "epoch": 0.233550979407333, "loss": 0.0029, "grad_norm": 5.265488147735596, "learning_rate": 2.957575757575758e-06, "num_tokens": 4288974.0, "completions/mean_length": 137.375, "completions/min_length": 124.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 137.375, "completions/min_terminated_length": 124.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.9915063381195068, "rewards/meter/std": 0.0047904085367918015, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9188686013221741, "rewards/repeat_soft/std": 0.04687410593032837, "rewards/judge_quality/mean": 0.24249999225139618, "rewards/judge_quality/std": 0.11792854964733124, "rewards/total_composite/mean": 0.7608146667480469, "rewards/total_composite/std": 0.033390771597623825, "reward": 0.7608146667480469, "reward_std": 0.03339076787233353, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10047654062509537, "sampling/sampling_logp_difference/max": 1.4179396629333496, "sampling/importance_sampling_ratio/min": 0.24221254885196686, "sampling/importance_sampling_ratio/mean": 1.0111217498779297, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6860749050974846, "clip_ratio/low_mean": 0.06159132719039917, "clip_ratio/low_min": 0.06159132719039917, "clip_ratio/high_mean": 0.038132776506245136, "clip_ratio/high_max": 0.038132776506245136, "clip_ratio/region_mean": 0.0997241036966443, "reward_total_mean": 0.7608146667480469, "reward_meter_mean": 0.9915063381195068, "reward_meter_std": 0.0047904085367918015, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9188686013221741, "reward_repeat_soft_std": 0.04687410593032837, "reward_judge_quality_mean": 0.24249999225139618, "reward_judge_quality_std": 0.11792854964733124, "reward_total_composite_mean": 0.7608146667480469, "reward_total_composite_std": 0.033390771597623825} {"timestamp_utc": "2026-04-13T04:10:15Z", "mode": "train", "global_step": 2326, "epoch": 0.2336514314414867, "loss": 0.0059, "grad_norm": 5.896433353424072, "learning_rate": 2.954545454545455e-06, "num_tokens": 4291591.0, "completions/mean_length": 142.125, "completions/min_length": 136.0, "completions/max_length": 150.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.125, "completions/min_terminated_length": 136.0, "completions/max_terminated_length": 150.0, "rewards/meter/mean": 0.9180581569671631, "rewards/meter/std": 0.11271790415048599, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9653102159500122, "rewards/repeat_soft/std": 0.019736677408218384, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.7755321860313416, "rewards/total_composite/std": 0.0527278408408165, "reward": 0.7755321860313416, "reward_std": 0.052727848291397095, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09958676248788834, "sampling/sampling_logp_difference/max": 2.061572313308716, "sampling/importance_sampling_ratio/min": 0.12725374102592468, "sampling/importance_sampling_ratio/mean": 1.0027756690979004, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5415206104516983, "clip_ratio/low_mean": 0.037001051008701324, "clip_ratio/low_min": 0.037001051008701324, "clip_ratio/high_mean": 0.07551316544413567, "clip_ratio/high_max": 0.07551316544413567, "clip_ratio/region_mean": 0.11251421645283699, "reward_total_mean": 0.7755321860313416, "reward_meter_mean": 0.9180581569671631, "reward_meter_std": 0.11271790415048599, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9653102159500122, "reward_repeat_soft_std": 0.019736677408218384, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.7755321860313416, "reward_total_composite_std": 0.0527278408408165} {"timestamp_utc": "2026-04-13T04:10:22Z", "mode": "train", "global_step": 2327, "epoch": 0.23375188347564038, "loss": 0.0407, "grad_norm": 8.850568771362305, "learning_rate": 2.951515151515152e-06, "num_tokens": 4293407.0, "completions/mean_length": 73.0, "completions/min_length": 67.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9732934236526489, "rewards/meter/std": 0.028120307251811028, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9803524017333984, "rewards/repeat_soft/std": 0.01894899643957615, "rewards/judge_quality/mean": 0.5862500667572021, "rewards/judge_quality/std": 0.2822834253311157, "rewards/total_composite/mean": 0.8618923425674438, "rewards/total_composite/std": 0.07966319471597672, "reward": 0.8618923425674438, "reward_std": 0.07966319471597672, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12195608019828796, "sampling/sampling_logp_difference/max": 1.5411067008972168, "sampling/importance_sampling_ratio/min": 0.21414399147033691, "sampling/importance_sampling_ratio/mean": 1.0300182104110718, "sampling/importance_sampling_ratio/max": 1.9737838506698608, "entropy": 0.7862537950277328, "clip_ratio/low_mean": 0.08358898293226957, "clip_ratio/low_min": 0.08358898293226957, "clip_ratio/high_mean": 0.03413305152207613, "clip_ratio/high_max": 0.03413305152207613, "clip_ratio/region_mean": 0.1177220344543457, "reward_total_mean": 0.8618923425674438, "reward_meter_mean": 0.9732934236526489, "reward_meter_std": 0.028120307251811028, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9803524017333984, "reward_repeat_soft_std": 0.01894899643957615, "reward_judge_quality_mean": 0.5862500667572021, "reward_judge_quality_std": 0.2822834253311157, "reward_total_composite_mean": 0.8618923425674438, "reward_total_composite_std": 0.07966319471597672} {"timestamp_utc": "2026-04-13T04:10:29Z", "mode": "train", "global_step": 2328, "epoch": 0.23385233550979406, "loss": 0.0224, "grad_norm": 9.319945335388184, "learning_rate": 2.9484848484848488e-06, "num_tokens": 4295472.0, "completions/mean_length": 89.125, "completions/min_length": 83.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.125, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.876101553440094, "rewards/meter/std": 0.28966835141181946, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9615591764450073, "rewards/repeat_soft/std": 0.04372605308890343, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.20860080420970917, "rewards/total_composite/mean": 0.7724016308784485, "rewards/total_composite/std": 0.14513041079044342, "reward": 0.7724016308784485, "reward_std": 0.14513041079044342, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10596959292888641, "sampling/sampling_logp_difference/max": 1.6589584350585938, "sampling/importance_sampling_ratio/min": 0.190337136387825, "sampling/importance_sampling_ratio/mean": 1.0030089616775513, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6572568938136101, "clip_ratio/low_mean": 0.042917389422655106, "clip_ratio/low_min": 0.042917389422655106, "clip_ratio/high_mean": 0.06257760524749756, "clip_ratio/high_max": 0.06257760524749756, "clip_ratio/region_mean": 0.10549499467015266, "reward_total_mean": 0.7724016308784485, "reward_meter_mean": 0.876101553440094, "reward_meter_std": 0.28966835141181946, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9615591764450073, "reward_repeat_soft_std": 0.04372605308890343, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.20860080420970917, "reward_total_composite_mean": 0.7724016308784485, "reward_total_composite_std": 0.14513041079044342} {"timestamp_utc": "2026-04-13T04:10:36Z", "mode": "train", "global_step": 2329, "epoch": 0.23395278754394777, "loss": 0.0502, "grad_norm": 6.231706619262695, "learning_rate": 2.9454545454545456e-06, "num_tokens": 4297571.0, "completions/mean_length": 106.375, "completions/min_length": 92.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.375, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.9792406558990479, "rewards/meter/std": 0.00814460776746273, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9089250564575195, "rewards/repeat_soft/std": 0.04953783005475998, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.7783007621765137, "rewards/total_composite/std": 0.034515343606472015, "reward": 0.7783007621765137, "reward_std": 0.03451536223292351, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10637635737657547, "sampling/sampling_logp_difference/max": 2.3236353397369385, "sampling/importance_sampling_ratio/min": 0.09791697561740875, "sampling/importance_sampling_ratio/mean": 1.0011701583862305, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.672793060541153, "clip_ratio/low_mean": 0.051209389697760344, "clip_ratio/low_min": 0.051209389697760344, "clip_ratio/high_mean": 0.05478856526315212, "clip_ratio/high_max": 0.05478856526315212, "clip_ratio/region_mean": 0.10599795496091247, "reward_total_mean": 0.7783007621765137, "reward_meter_mean": 0.9792406558990479, "reward_meter_std": 0.00814460776746273, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9089250564575195, "reward_repeat_soft_std": 0.04953783005475998, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.7783007621765137, "reward_total_composite_std": 0.034515343606472015} {"timestamp_utc": "2026-04-13T04:10:43Z", "mode": "train", "global_step": 2330, "epoch": 0.23405323957810145, "loss": -0.0107, "grad_norm": 9.475613594055176, "learning_rate": 2.942424242424243e-06, "num_tokens": 4299150.0, "completions/mean_length": 41.375, "completions/min_length": 38.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.375, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.9332320690155029, "rewards/meter/std": 0.01930769346654415, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9489104747772217, "rewards/repeat_soft/std": 0.04937707632780075, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.7889704704284668, "rewards/total_composite/std": 0.01591862551867962, "reward": 0.7889704704284668, "reward_std": 0.015918616205453873, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08379632234573364, "sampling/sampling_logp_difference/max": 1.498443603515625, "sampling/importance_sampling_ratio/min": 0.22347772121429443, "sampling/importance_sampling_ratio/mean": 1.0101401805877686, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45587772130966187, "clip_ratio/low_mean": 0.03037839848548174, "clip_ratio/low_min": 0.03037839848548174, "clip_ratio/high_mean": 0.03287232341244817, "clip_ratio/high_max": 0.03287232341244817, "clip_ratio/region_mean": 0.0632507218979299, "reward_total_mean": 0.7889704704284668, "reward_meter_mean": 0.9332320690155029, "reward_meter_std": 0.01930769346654415, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9489104747772217, "reward_repeat_soft_std": 0.04937707632780075, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.7889704704284668, "reward_total_composite_std": 0.01591862551867962} {"timestamp_utc": "2026-04-13T04:10:49Z", "mode": "train", "global_step": 2331, "epoch": 0.23415369161225516, "loss": 0.0203, "grad_norm": 12.546533584594727, "learning_rate": 2.9393939393939397e-06, "num_tokens": 4300841.0, "completions/mean_length": 56.375, "completions/min_length": 48.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.375, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9677978157997131, "rewards/meter/std": 0.03629437834024429, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9971694946289062, "rewards/repeat_soft/std": 0.002741617849096656, "rewards/judge_quality/mean": 0.3999999761581421, "rewards/judge_quality/std": 0.09258200973272324, "rewards/total_composite/mean": 0.8052259683609009, "rewards/total_composite/std": 0.03605028614401817, "reward": 0.8052259683609009, "reward_std": 0.03605028986930847, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11690624058246613, "sampling/sampling_logp_difference/max": 1.0705211162567139, "sampling/importance_sampling_ratio/min": 0.3428298234939575, "sampling/importance_sampling_ratio/mean": 1.015939474105835, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8159402459859848, "clip_ratio/low_mean": 0.06699388939887285, "clip_ratio/low_min": 0.06699388939887285, "clip_ratio/high_mean": 0.06665081065148115, "clip_ratio/high_max": 0.06665081065148115, "clip_ratio/region_mean": 0.133644700050354, "reward_total_mean": 0.8052259683609009, "reward_meter_mean": 0.9677978157997131, "reward_meter_std": 0.03629437834024429, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9971694946289062, "reward_repeat_soft_std": 0.002741617849096656, "reward_judge_quality_mean": 0.3999999761581421, "reward_judge_quality_std": 0.09258200973272324, "reward_total_composite_mean": 0.8052259683609009, "reward_total_composite_std": 0.03605028614401817} {"timestamp_utc": "2026-04-13T04:10:57Z", "mode": "train", "global_step": 2332, "epoch": 0.23425414364640884, "loss": 0.0312, "grad_norm": 6.980352401733398, "learning_rate": 2.9363636363636365e-06, "num_tokens": 4303166.0, "completions/mean_length": 124.625, "completions/min_length": 105.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 124.625, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.9850785732269287, "rewards/meter/std": 0.006025006528943777, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8221598863601685, "rewards/repeat_soft/std": 0.08202683180570602, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7823763489723206, "rewards/total_composite/std": 0.02813875302672386, "reward": 0.7823763489723206, "reward_std": 0.028138745576143265, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08056610077619553, "sampling/sampling_logp_difference/max": 1.785904884338379, "sampling/importance_sampling_ratio/min": 0.16764529049396515, "sampling/importance_sampling_ratio/mean": 1.0075500011444092, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43925778567790985, "clip_ratio/low_mean": 0.036787777207791805, "clip_ratio/low_min": 0.036787777207791805, "clip_ratio/high_mean": 0.04581672535277903, "clip_ratio/high_max": 0.04581672535277903, "clip_ratio/region_mean": 0.08260450256057084, "reward_total_mean": 0.7823763489723206, "reward_meter_mean": 0.9850785732269287, "reward_meter_std": 0.006025006528943777, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8221598863601685, "reward_repeat_soft_std": 0.08202683180570602, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7823763489723206, "reward_total_composite_std": 0.02813875302672386} {"timestamp_utc": "2026-04-13T04:11:09Z", "mode": "train", "global_step": 2333, "epoch": 0.23435459568056252, "loss": -0.0922, "grad_norm": 2.4124810695648193, "learning_rate": 2.9333333333333338e-06, "num_tokens": 4304589.0, "completions/mean_length": 90.875, "completions/min_length": 29.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 30.71428680419922, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.7049896121025085, "rewards/meter/std": 0.40509745478630066, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.961480975151062, "rewards/repeat_soft/std": 0.02230331115424633, "rewards/judge_quality/mean": 0.815000057220459, "rewards/judge_quality/std": 0.30928489565849304, "rewards/total_composite/mean": 0.7747684121131897, "rewards/total_composite/std": 0.33924785256385803, "reward": 0.7747684121131897, "reward_std": 0.33924785256385803, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1171395406126976, "sampling/sampling_logp_difference/max": 1.0211944580078125, "sampling/importance_sampling_ratio/min": 0.36016449332237244, "sampling/importance_sampling_ratio/mean": 1.0112260580062866, "sampling/importance_sampling_ratio/max": 1.9200782775878906, "entropy": 0.5569302961230278, "clip_ratio/low_mean": 0.012500000186264515, "clip_ratio/low_min": 0.012500000186264515, "clip_ratio/high_mean": 0.08184226043522358, "clip_ratio/high_max": 0.08184226043522358, "clip_ratio/region_mean": 0.0943422606214881, "reward_total_mean": 0.7747684121131897, "reward_meter_mean": 0.7049896121025085, "reward_meter_std": 0.40509745478630066, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.961480975151062, "reward_repeat_soft_std": 0.02230331115424633, "reward_judge_quality_mean": 0.815000057220459, "reward_judge_quality_std": 0.30928489565849304, "reward_total_composite_mean": 0.7747684121131897, "reward_total_composite_std": 0.33924785256385803} {"timestamp_utc": "2026-04-13T04:11:20Z", "mode": "train", "global_step": 2334, "epoch": 0.23445504771471623, "loss": -0.1425, "grad_norm": 2.8984317779541016, "learning_rate": 2.9303030303030306e-06, "num_tokens": 4306321.0, "completions/mean_length": 123.5, "completions/min_length": 62.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.6892333626747131, "rewards/meter/std": 0.3999255895614624, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9833346605300903, "rewards/repeat_soft/std": 0.01741868630051613, "rewards/judge_quality/mean": 0.29750001430511475, "rewards/judge_quality/std": 0.14518460631370544, "rewards/total_composite/mean": 0.6059813499450684, "rewards/total_composite/std": 0.29569005966186523, "reward": 0.6059813499450684, "reward_std": 0.29569005966186523, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11773686856031418, "sampling/sampling_logp_difference/max": 1.3770743608474731, "sampling/importance_sampling_ratio/min": 0.2523156702518463, "sampling/importance_sampling_ratio/mean": 1.0017814636230469, "sampling/importance_sampling_ratio/max": 1.8436490297317505, "entropy": 0.6461654976010323, "clip_ratio/low_mean": 0.03684371244162321, "clip_ratio/low_min": 0.03684371244162321, "clip_ratio/high_mean": 0.0800736784003675, "clip_ratio/high_max": 0.0800736784003675, "clip_ratio/region_mean": 0.11691739084199071, "reward_total_mean": 0.6059813499450684, "reward_meter_mean": 0.6892333626747131, "reward_meter_std": 0.3999255895614624, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9833346605300903, "reward_repeat_soft_std": 0.01741868630051613, "reward_judge_quality_mean": 0.29750001430511475, "reward_judge_quality_std": 0.14518460631370544, "reward_total_composite_mean": 0.6059813499450684, "reward_total_composite_std": 0.29569005966186523} {"timestamp_utc": "2026-04-13T04:11:33Z", "mode": "train", "global_step": 2335, "epoch": 0.2345554997488699, "loss": 0.0376, "grad_norm": 11.131345748901367, "learning_rate": 2.9272727272727274e-06, "num_tokens": 4308065.0, "completions/mean_length": 57.0, "completions/min_length": 52.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.0, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.8678208589553833, "rewards/meter/std": 0.312097430229187, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9717869162559509, "rewards/repeat_soft/std": 0.029565274715423584, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.8386980891227722, "rewards/total_composite/std": 0.13333196938037872, "reward": 0.8386980891227722, "reward_std": 0.13333195447921753, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10661342740058899, "sampling/sampling_logp_difference/max": 1.2401357889175415, "sampling/importance_sampling_ratio/min": 0.2893449366092682, "sampling/importance_sampling_ratio/mean": 0.9988555908203125, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6812417209148407, "clip_ratio/low_mean": 0.05872205412015319, "clip_ratio/low_min": 0.05872205412015319, "clip_ratio/high_mean": 0.053218984976410866, "clip_ratio/high_max": 0.053218984976410866, "clip_ratio/region_mean": 0.11194103909656405, "reward_total_mean": 0.8386980891227722, "reward_meter_mean": 0.8678208589553833, "reward_meter_std": 0.312097430229187, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9717869162559509, "reward_repeat_soft_std": 0.029565274715423584, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.8386980891227722, "reward_total_composite_std": 0.13333196938037872} {"timestamp_utc": "2026-04-13T04:11:45Z", "mode": "train", "global_step": 2336, "epoch": 0.23465595178302362, "loss": -0.2494, "grad_norm": 1.829311490058899, "learning_rate": 2.9242424242424243e-06, "num_tokens": 4311245.0, "completions/mean_length": 249.5, "completions/min_length": 200.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 212.00001525878906, "completions/min_terminated_length": 200.0, "completions/max_terminated_length": 218.0, "rewards/meter/mean": 0.8599675893783569, "rewards/meter/std": 0.2278381884098053, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9181342124938965, "rewards/repeat_soft/std": 0.07190245389938354, "rewards/judge_quality/mean": 0.2475000023841858, "rewards/judge_quality/std": 0.14320313930511475, "rewards/total_composite/mean": 0.6442697644233704, "rewards/total_composite/std": 0.27386489510536194, "reward": 0.6442697644233704, "reward_std": 0.27386486530303955, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08602063357830048, "sampling/sampling_logp_difference/max": 1.7187299728393555, "sampling/importance_sampling_ratio/min": 0.1792937070131302, "sampling/importance_sampling_ratio/mean": 1.011961579322815, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5396718233823776, "clip_ratio/low_mean": 0.008971291594207287, "clip_ratio/low_min": 0.008971291594207287, "clip_ratio/high_mean": 0.06781934527680278, "clip_ratio/high_max": 0.06781934527680278, "clip_ratio/region_mean": 0.07679063687101007, "reward_total_mean": 0.6442697644233704, "reward_meter_mean": 0.8599675893783569, "reward_meter_std": 0.2278381884098053, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9181342124938965, "reward_repeat_soft_std": 0.07190245389938354, "reward_judge_quality_mean": 0.2475000023841858, "reward_judge_quality_std": 0.14320313930511475, "reward_total_composite_mean": 0.6442697644233704, "reward_total_composite_std": 0.27386489510536194} {"timestamp_utc": "2026-04-13T04:11:51Z", "mode": "train", "global_step": 2337, "epoch": 0.2347564038171773, "loss": 0.0621, "grad_norm": 23.849679946899414, "learning_rate": 2.9212121212121215e-06, "num_tokens": 4312748.0, "completions/mean_length": 43.875, "completions/min_length": 41.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.875, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.776848554611206, "rewards/meter/std": 0.3188053369522095, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9693849086761475, "rewards/repeat_soft/std": 0.04004884511232376, "rewards/judge_quality/mean": 0.4649999737739563, "rewards/judge_quality/std": 0.10392307490110397, "rewards/total_composite/mean": 0.7360203266143799, "rewards/total_composite/std": 0.13494479656219482, "reward": 0.7360203266143799, "reward_std": 0.13494479656219482, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1053723469376564, "sampling/sampling_logp_difference/max": 1.8853201866149902, "sampling/importance_sampling_ratio/min": 0.15178045630455017, "sampling/importance_sampling_ratio/mean": 1.0162804126739502, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6173879280686378, "clip_ratio/low_mean": 0.017309725284576416, "clip_ratio/low_min": 0.017309725284576416, "clip_ratio/high_mean": 0.06628607492893934, "clip_ratio/high_max": 0.06628607492893934, "clip_ratio/region_mean": 0.08359580021351576, "reward_total_mean": 0.7360203266143799, "reward_meter_mean": 0.776848554611206, "reward_meter_std": 0.3188053369522095, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9693849086761475, "reward_repeat_soft_std": 0.04004884511232376, "reward_judge_quality_mean": 0.4649999737739563, "reward_judge_quality_std": 0.10392307490110397, "reward_total_composite_mean": 0.7360203266143799, "reward_total_composite_std": 0.13494479656219482} {"timestamp_utc": "2026-04-13T04:12:06Z", "mode": "train", "global_step": 2338, "epoch": 0.23485685585133098, "loss": -0.1083, "grad_norm": 2.6215670108795166, "learning_rate": 2.9181818181818183e-06, "num_tokens": 4314291.0, "completions/mean_length": 103.875, "completions/min_length": 39.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 45.57143020629883, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.6460777521133423, "rewards/meter/std": 0.3965824544429779, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9626538753509521, "rewards/repeat_soft/std": 0.041867583990097046, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.1566559076309204, "rewards/total_composite/mean": 0.6063754558563232, "rewards/total_composite/std": 0.28200438618659973, "reward": 0.6063754558563232, "reward_std": 0.28200438618659973, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09168768674135208, "sampling/sampling_logp_difference/max": 1.218064308166504, "sampling/importance_sampling_ratio/min": 0.29580220580101013, "sampling/importance_sampling_ratio/mean": 1.0098472833633423, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4704268164932728, "clip_ratio/low_mean": 0.016188714187592268, "clip_ratio/low_min": 0.016188714187592268, "clip_ratio/high_mean": 0.05394681403413415, "clip_ratio/high_max": 0.05394681403413415, "clip_ratio/region_mean": 0.07013552822172642, "reward_total_mean": 0.6063754558563232, "reward_meter_mean": 0.6460777521133423, "reward_meter_std": 0.3965824544429779, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9626538753509521, "reward_repeat_soft_std": 0.041867583990097046, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.1566559076309204, "reward_total_composite_mean": 0.6063754558563232, "reward_total_composite_std": 0.28200438618659973} {"timestamp_utc": "2026-04-13T04:12:14Z", "mode": "train", "global_step": 2339, "epoch": 0.2349573078854847, "loss": 0.0163, "grad_norm": 6.2456746101379395, "learning_rate": 2.915151515151515e-06, "num_tokens": 4316986.0, "completions/mean_length": 156.875, "completions/min_length": 141.0, "completions/max_length": 173.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 156.875, "completions/min_terminated_length": 141.0, "completions/max_terminated_length": 173.0, "rewards/meter/mean": 0.7010831236839294, "rewards/meter/std": 0.15589667856693268, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9445762634277344, "rewards/repeat_soft/std": 0.023395059630274773, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.6529450416564941, "rewards/total_composite/std": 0.09649792313575745, "reward": 0.6529450416564941, "reward_std": 0.09649790823459625, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1034923791885376, "sampling/sampling_logp_difference/max": 1.8211865425109863, "sampling/importance_sampling_ratio/min": 0.16183361411094666, "sampling/importance_sampling_ratio/mean": 1.008095383644104, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5806264132261276, "clip_ratio/low_mean": 0.04935429897159338, "clip_ratio/low_min": 0.04935429897159338, "clip_ratio/high_mean": 0.050342999398708344, "clip_ratio/high_max": 0.050342999398708344, "clip_ratio/region_mean": 0.09969729837030172, "reward_total_mean": 0.6529450416564941, "reward_meter_mean": 0.7010831236839294, "reward_meter_std": 0.15589667856693268, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9445762634277344, "reward_repeat_soft_std": 0.023395059630274773, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.6529450416564941, "reward_total_composite_std": 0.09649792313575745} {"timestamp_utc": "2026-04-13T04:12:22Z", "mode": "train", "global_step": 2340, "epoch": 0.23505775991963837, "loss": 0.0415, "grad_norm": 4.808964252471924, "learning_rate": 2.9121212121212124e-06, "num_tokens": 4319776.0, "completions/mean_length": 171.75, "completions/min_length": 161.0, "completions/max_length": 182.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 171.75, "completions/min_terminated_length": 161.0, "completions/max_terminated_length": 182.0, "rewards/meter/mean": 0.9680657386779785, "rewards/meter/std": 0.05579310283064842, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9077001810073853, "rewards/repeat_soft/std": 0.05540850758552551, "rewards/judge_quality/mean": 0.26749998331069946, "rewards/judge_quality/std": 0.10375107079744339, "rewards/total_composite/mean": 0.7566496133804321, "rewards/total_composite/std": 0.027886822819709778, "reward": 0.7566496133804321, "reward_std": 0.027886833995580673, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10274724662303925, "sampling/sampling_logp_difference/max": 2.317783832550049, "sampling/importance_sampling_ratio/min": 0.0984916239976883, "sampling/importance_sampling_ratio/mean": 1.0179109573364258, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.724169485270977, "clip_ratio/low_mean": 0.02988720592111349, "clip_ratio/low_min": 0.02988720592111349, "clip_ratio/high_mean": 0.05713123083114624, "clip_ratio/high_max": 0.05713123083114624, "clip_ratio/region_mean": 0.08701843675225973, "reward_total_mean": 0.7566496133804321, "reward_meter_mean": 0.9680657386779785, "reward_meter_std": 0.05579310283064842, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9077001810073853, "reward_repeat_soft_std": 0.05540850758552551, "reward_judge_quality_mean": 0.26749998331069946, "reward_judge_quality_std": 0.10375107079744339, "reward_total_composite_mean": 0.7566496133804321, "reward_total_composite_std": 0.027886822819709778} {"timestamp_utc": "2026-04-13T04:12:30Z", "mode": "train", "global_step": 2341, "epoch": 0.23515821195379205, "loss": -0.0124, "grad_norm": 4.482976913452148, "learning_rate": 2.9090909090909093e-06, "num_tokens": 4322749.0, "completions/mean_length": 176.625, "completions/min_length": 151.0, "completions/max_length": 193.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 176.625, "completions/min_terminated_length": 151.0, "completions/max_terminated_length": 193.0, "rewards/meter/mean": 0.9948548078536987, "rewards/meter/std": 0.0014725212240591645, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.07715168595314026, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9141745567321777, "rewards/repeat_soft/std": 0.05563684552907944, "rewards/judge_quality/mean": 0.3462499976158142, "rewards/judge_quality/std": 0.2392212152481079, "rewards/total_composite/mean": 0.7742271423339844, "rewards/total_composite/std": 0.07995603233575821, "reward": 0.7742271423339844, "reward_std": 0.07995601743459702, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09840737283229828, "sampling/sampling_logp_difference/max": 3.1393356323242188, "sampling/importance_sampling_ratio/min": 0.04331156238913536, "sampling/importance_sampling_ratio/mean": 1.007741928100586, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.580645527690649, "clip_ratio/low_mean": 0.05249261204153299, "clip_ratio/low_min": 0.05249261204153299, "clip_ratio/high_mean": 0.03331385925412178, "clip_ratio/high_max": 0.03331385925412178, "clip_ratio/region_mean": 0.08580647129565477, "reward_total_mean": 0.7742271423339844, "reward_meter_mean": 0.9948548078536987, "reward_meter_std": 0.0014725212240591645, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.07715168595314026, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9141745567321777, "reward_repeat_soft_std": 0.05563684552907944, "reward_judge_quality_mean": 0.3462499976158142, "reward_judge_quality_std": 0.2392212152481079, "reward_total_composite_mean": 0.7742271423339844, "reward_total_composite_std": 0.07995603233575821} {"timestamp_utc": "2026-04-13T04:12:42Z", "mode": "train", "global_step": 2342, "epoch": 0.23525866398794576, "loss": 0.0123, "grad_norm": 6.805415630340576, "learning_rate": 2.906060606060606e-06, "num_tokens": 4324933.0, "completions/mean_length": 114.0, "completions/min_length": 108.0, "completions/max_length": 127.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.0, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.8840720653533936, "rewards/meter/std": 0.13761039078235626, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.960912823677063, "rewards/repeat_soft/std": 0.031703222543001175, "rewards/judge_quality/mean": 0.2887499928474426, "rewards/judge_quality/std": 0.11630470305681229, "rewards/total_composite/mean": 0.7305487394332886, "rewards/total_composite/std": 0.07739962637424469, "reward": 0.7305487394332886, "reward_std": 0.07739961892366409, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1155419796705246, "sampling/sampling_logp_difference/max": 1.6248579025268555, "sampling/importance_sampling_ratio/min": 0.19693966209888458, "sampling/importance_sampling_ratio/mean": 1.0038461685180664, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8788465782999992, "clip_ratio/low_mean": 0.05832547787576914, "clip_ratio/low_min": 0.05832547787576914, "clip_ratio/high_mean": 0.07151516992598772, "clip_ratio/high_max": 0.07151516992598772, "clip_ratio/region_mean": 0.12984064780175686, "reward_total_mean": 0.7305487394332886, "reward_meter_mean": 0.8840720653533936, "reward_meter_std": 0.13761039078235626, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.960912823677063, "reward_repeat_soft_std": 0.031703222543001175, "reward_judge_quality_mean": 0.2887499928474426, "reward_judge_quality_std": 0.11630470305681229, "reward_total_composite_mean": 0.7305487394332886, "reward_total_composite_std": 0.07739962637424469} {"timestamp_utc": "2026-04-13T04:12:49Z", "mode": "train", "global_step": 2343, "epoch": 0.23535911602209944, "loss": 0.0239, "grad_norm": 13.213395118713379, "learning_rate": 2.903030303030303e-06, "num_tokens": 4326971.0, "completions/mean_length": 85.75, "completions/min_length": 69.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.75, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.8652454614639282, "rewards/meter/std": 0.23155654966831207, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.94980788230896, "rewards/repeat_soft/std": 0.0381682850420475, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.0975411981344223, "rewards/total_composite/mean": 0.7466537356376648, "rewards/total_composite/std": 0.09876105189323425, "reward": 0.7466537356376648, "reward_std": 0.09876106679439545, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15276256203651428, "sampling/sampling_logp_difference/max": 2.2917823791503906, "sampling/importance_sampling_ratio/min": 0.10108613222837448, "sampling/importance_sampling_ratio/mean": 0.9952856302261353, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5597515106201172, "clip_ratio/low_mean": 0.04507553856819868, "clip_ratio/low_min": 0.04507553856819868, "clip_ratio/high_mean": 0.08575550559908152, "clip_ratio/high_max": 0.08575550559908152, "clip_ratio/region_mean": 0.1308310441672802, "reward_total_mean": 0.7466537356376648, "reward_meter_mean": 0.8652454614639282, "reward_meter_std": 0.23155654966831207, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.94980788230896, "reward_repeat_soft_std": 0.0381682850420475, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.0975411981344223, "reward_total_composite_mean": 0.7466537356376648, "reward_total_composite_std": 0.09876105189323425} {"timestamp_utc": "2026-04-13T04:12:55Z", "mode": "train", "global_step": 2344, "epoch": 0.23545956805625315, "loss": 0.0027, "grad_norm": 13.860200881958008, "learning_rate": 2.9e-06, "num_tokens": 4328600.0, "completions/mean_length": 50.625, "completions/min_length": 41.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.625, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9844867587089539, "rewards/meter/std": 0.025858668610453606, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9753512144088745, "rewards/repeat_soft/std": 0.04901522397994995, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.8728041648864746, "rewards/total_composite/std": 0.08419115096330643, "reward": 0.8728041648864746, "reward_std": 0.08419115096330643, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13038958609104156, "sampling/sampling_logp_difference/max": 2.0022754669189453, "sampling/importance_sampling_ratio/min": 0.135027676820755, "sampling/importance_sampling_ratio/mean": 1.0212736129760742, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.731669545173645, "clip_ratio/low_mean": 0.07804703805595636, "clip_ratio/low_min": 0.07804703805595636, "clip_ratio/high_mean": 0.03777777682989836, "clip_ratio/high_max": 0.03777777682989836, "clip_ratio/region_mean": 0.11582481488585472, "reward_total_mean": 0.8728041648864746, "reward_meter_mean": 0.9844867587089539, "reward_meter_std": 0.025858668610453606, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9753512144088745, "reward_repeat_soft_std": 0.04901522397994995, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.8728041648864746, "reward_total_composite_std": 0.08419115096330643} {"timestamp_utc": "2026-04-13T04:13:07Z", "mode": "train", "global_step": 2345, "epoch": 0.23556002009040683, "loss": -0.0979, "grad_norm": 3.3104870319366455, "learning_rate": 2.896969696969697e-06, "num_tokens": 4330201.0, "completions/mean_length": 107.125, "completions/min_length": 40.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 49.28571701049805, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.3448673486709595, "rewards/meter/std": 0.39002248644828796, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9816499352455139, "rewards/repeat_soft/std": 0.022002840414643288, "rewards/judge_quality/mean": 0.7487499713897705, "rewards/judge_quality/std": 0.332154780626297, "rewards/total_composite/mean": 0.5415583848953247, "rewards/total_composite/std": 0.27290788292884827, "reward": 0.5415583848953247, "reward_std": 0.27290788292884827, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12351860851049423, "sampling/sampling_logp_difference/max": 1.6155171394348145, "sampling/importance_sampling_ratio/min": 0.1987878382205963, "sampling/importance_sampling_ratio/mean": 0.9962128400802612, "sampling/importance_sampling_ratio/max": 1.7513020038604736, "entropy": 0.6885268315672874, "clip_ratio/low_mean": 0.022557131946086884, "clip_ratio/low_min": 0.022557131946086884, "clip_ratio/high_mean": 0.10448916535824537, "clip_ratio/high_max": 0.10448916535824537, "clip_ratio/region_mean": 0.12704629730433226, "reward_total_mean": 0.5415583848953247, "reward_meter_mean": 0.3448673486709595, "reward_meter_std": 0.39002248644828796, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9816499352455139, "reward_repeat_soft_std": 0.022002840414643288, "reward_judge_quality_mean": 0.7487499713897705, "reward_judge_quality_std": 0.332154780626297, "reward_total_composite_mean": 0.5415583848953247, "reward_total_composite_std": 0.27290788292884827} {"timestamp_utc": "2026-04-13T04:13:13Z", "mode": "train", "global_step": 2346, "epoch": 0.2356604721245605, "loss": 0.1043, "grad_norm": 12.762657165527344, "learning_rate": 2.893939393939394e-06, "num_tokens": 4331852.0, "completions/mean_length": 46.375, "completions/min_length": 37.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.375, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.5928311347961426, "rewards/meter/std": 0.3457811772823334, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9918125867843628, "rewards/repeat_soft/std": 0.013348426669836044, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.6640802621841431, "rewards/total_composite/std": 0.18342328071594238, "reward": 0.6640802621841431, "reward_std": 0.1834232658147812, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15094228088855743, "sampling/sampling_logp_difference/max": 1.1806755065917969, "sampling/importance_sampling_ratio/min": 0.3070712387561798, "sampling/importance_sampling_ratio/mean": 1.0304207801818848, "sampling/importance_sampling_ratio/max": 1.8144605159759521, "entropy": 1.1602900847792625, "clip_ratio/low_mean": 0.08004794735461473, "clip_ratio/low_min": 0.08004794735461473, "clip_ratio/high_mean": 0.0792837580665946, "clip_ratio/high_max": 0.0792837580665946, "clip_ratio/region_mean": 0.15933170542120934, "reward_total_mean": 0.6640802621841431, "reward_meter_mean": 0.5928311347961426, "reward_meter_std": 0.3457811772823334, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9918125867843628, "reward_repeat_soft_std": 0.013348426669836044, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.6640802621841431, "reward_total_composite_std": 0.18342328071594238} {"timestamp_utc": "2026-04-13T04:13:25Z", "mode": "train", "global_step": 2347, "epoch": 0.23576092415871422, "loss": -0.1076, "grad_norm": 1.2071073055267334, "learning_rate": 2.8909090909090907e-06, "num_tokens": 4333284.0, "completions/mean_length": 93.0, "completions/min_length": 30.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 33.142860412597656, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.9807363748550415, "rewards/meter/std": 0.024332968518137932, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9386353492736816, "rewards/repeat_soft/std": 0.06465566903352737, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.13617216050624847, "rewards/total_composite/mean": 0.7160260081291199, "rewards/total_composite/std": 0.28940749168395996, "reward": 0.7160260081291199, "reward_std": 0.28940749168395996, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09341710060834885, "sampling/sampling_logp_difference/max": 0.876460075378418, "sampling/importance_sampling_ratio/min": 0.41625380516052246, "sampling/importance_sampling_ratio/mean": 1.0110852718353271, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4843719154596329, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10993329994380474, "clip_ratio/high_max": 0.10993329994380474, "clip_ratio/region_mean": 0.10993329994380474, "reward_total_mean": 0.7160260081291199, "reward_meter_mean": 0.9807363748550415, "reward_meter_std": 0.024332968518137932, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9386353492736816, "reward_repeat_soft_std": 0.06465566903352737, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.13617216050624847, "reward_total_composite_mean": 0.7160260081291199, "reward_total_composite_std": 0.28940749168395996} {"timestamp_utc": "2026-04-13T04:13:33Z", "mode": "train", "global_step": 2348, "epoch": 0.2358613761928679, "loss": -0.0147, "grad_norm": 5.559314250946045, "learning_rate": 2.8878787878787884e-06, "num_tokens": 4336040.0, "completions/mean_length": 142.5, "completions/min_length": 123.0, "completions/max_length": 155.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.5, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 155.0, "rewards/meter/mean": 0.9852714538574219, "rewards/meter/std": 0.005580480210483074, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7549546957015991, "rewards/repeat_soft/std": 0.10931439697742462, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.14520922303199768, "rewards/total_composite/mean": 0.7933676242828369, "rewards/total_composite/std": 0.044328659772872925, "reward": 0.7933676242828369, "reward_std": 0.044328656047582626, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07244051992893219, "sampling/sampling_logp_difference/max": 2.7839202880859375, "sampling/importance_sampling_ratio/min": 0.06179577484726906, "sampling/importance_sampling_ratio/mean": 1.0087350606918335, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.38217920809984207, "clip_ratio/low_mean": 0.03159250086173415, "clip_ratio/low_min": 0.03159250086173415, "clip_ratio/high_mean": 0.026716566178947687, "clip_ratio/high_max": 0.026716566178947687, "clip_ratio/region_mean": 0.05830906704068184, "reward_total_mean": 0.7933676242828369, "reward_meter_mean": 0.9852714538574219, "reward_meter_std": 0.005580480210483074, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7549546957015991, "reward_repeat_soft_std": 0.10931439697742462, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.14520922303199768, "reward_total_composite_mean": 0.7933676242828369, "reward_total_composite_std": 0.044328659772872925} {"timestamp_utc": "2026-04-13T04:13:41Z", "mode": "train", "global_step": 2349, "epoch": 0.2359618282270216, "loss": 0.0648, "grad_norm": 8.564059257507324, "learning_rate": 2.884848484848485e-06, "num_tokens": 4338029.0, "completions/mean_length": 81.625, "completions/min_length": 72.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.625, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.9741004705429077, "rewards/meter/std": 0.020237894728779793, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9605282545089722, "rewards/repeat_soft/std": 0.023373425006866455, "rewards/judge_quality/mean": 0.39375001192092896, "rewards/judge_quality/std": 0.09941794723272324, "rewards/total_composite/mean": 0.8025230169296265, "rewards/total_composite/std": 0.03691055253148079, "reward": 0.8025230169296265, "reward_std": 0.03691054880619049, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11846599727869034, "sampling/sampling_logp_difference/max": 3.259826183319092, "sampling/importance_sampling_ratio/min": 0.038395073264837265, "sampling/importance_sampling_ratio/mean": 1.0122549533843994, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6645439118146896, "clip_ratio/low_mean": 0.0027173913549631834, "clip_ratio/low_min": 0.0027173913549631834, "clip_ratio/high_mean": 0.09760508872568607, "clip_ratio/high_max": 0.09760508872568607, "clip_ratio/region_mean": 0.10032248008064926, "reward_total_mean": 0.8025230169296265, "reward_meter_mean": 0.9741004705429077, "reward_meter_std": 0.020237894728779793, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9605282545089722, "reward_repeat_soft_std": 0.023373425006866455, "reward_judge_quality_mean": 0.39375001192092896, "reward_judge_quality_std": 0.09941794723272324, "reward_total_composite_mean": 0.8025230169296265, "reward_total_composite_std": 0.03691055253148079} {"timestamp_utc": "2026-04-13T04:13:52Z", "mode": "train", "global_step": 2350, "epoch": 0.2360622802611753, "loss": -0.0989, "grad_norm": 2.2835581302642822, "learning_rate": 2.8818181818181824e-06, "num_tokens": 4339297.0, "completions/mean_length": 89.5, "completions/min_length": 25.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 29.142858505249023, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.6875351667404175, "rewards/meter/std": 0.34644150733947754, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.13617216050624847, "rewards/total_composite/mean": 0.6376814246177673, "rewards/total_composite/std": 0.2757343053817749, "reward": 0.6376814246177673, "reward_std": 0.2757343053817749, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10344835370779037, "sampling/sampling_logp_difference/max": 1.1175897121429443, "sampling/importance_sampling_ratio/min": 0.32706716656684875, "sampling/importance_sampling_ratio/mean": 1.0215989351272583, "sampling/importance_sampling_ratio/max": 1.656253457069397, "entropy": 0.6781466081738472, "clip_ratio/low_mean": 0.036551724188029766, "clip_ratio/low_min": 0.036551724188029766, "clip_ratio/high_mean": 0.059019139502197504, "clip_ratio/high_max": 0.059019139502197504, "clip_ratio/region_mean": 0.09557086369022727, "reward_total_mean": 0.6376814246177673, "reward_meter_mean": 0.6875351667404175, "reward_meter_std": 0.34644150733947754, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.13617216050624847, "reward_total_composite_mean": 0.6376814246177673, "reward_total_composite_std": 0.2757343053817749} {"timestamp_utc": "2026-04-13T04:14:48Z", "mode": "eval", "global_step": 2350, "epoch": 0.2360622802611753, "eval_loss": NaN, "eval_runtime": 55.8598, "eval_samples_per_second": 1.432, "eval_steps_per_second": 0.179, "eval_num_tokens": 4339297.0, "eval_completions/mean_length": 109.7625, "eval_completions/min_length": 42.6, "eval_completions/max_length": 241.0, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 99.13214340209962, "eval_completions/min_terminated_length": 42.6, "eval_completions/max_terminated_length": 169.7, "eval_rewards/meter/mean": 0.8899108648300171, "eval_rewards/meter/std": 0.14870471213944256, "eval_rewards/count_adherence/mean": 0.9572916686534881, "eval_rewards/count_adherence/std": 0.10326077528297901, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.07071067690849304, "eval_rewards/repeat_soft/mean": 0.9403779625892639, "eval_rewards/repeat_soft/std": 0.053413290344178674, "eval_rewards/judge_quality/mean": 0.3840000003576279, "eval_rewards/judge_quality/std": 0.15163438767194748, "eval_rewards/total_composite/mean": 0.7479026079177856, "eval_rewards/total_composite/std": 0.11637145765125752, "eval_reward": 0.7479026079177856, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.061882075667381284, "eval_sampling/sampling_logp_difference/max": 1.1607478618621827, "eval_sampling/importance_sampling_ratio/min": 0.3189630895853043, "eval_sampling/importance_sampling_ratio/mean": 1.012452447414398, "eval_sampling/importance_sampling_ratio/max": 1.4194721102714538, "eval_entropy": 0.6421197056770325, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7479026079177856, "eval_reward_meter_mean": 0.8899108648300171, "eval_reward_meter_std": 0.14870471213944256, "eval_reward_count_adherence_mean": 0.9572916686534881, "eval_reward_count_adherence_std": 0.10326077528297901, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.07071067690849304, "eval_reward_repeat_soft_mean": 0.9403779625892639, "eval_reward_repeat_soft_std": 0.053413290344178674, "eval_reward_judge_quality_mean": 0.3840000003576279, "eval_reward_judge_quality_std": 0.15163438767194748, "eval_reward_total_composite_mean": 0.7479026079177856, "eval_reward_total_composite_std": 0.11637145765125752} {"timestamp_utc": "2026-04-13T04:14:58Z", "mode": "train", "global_step": 2351, "epoch": 0.23616273229532897, "loss": 0.0381, "grad_norm": 8.74979019165039, "learning_rate": 2.8787878787878793e-06, "num_tokens": 4341191.0, "completions/mean_length": 67.75, "completions/min_length": 61.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.75, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.909159779548645, "rewards/meter/std": 0.07416226714849472, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9578614234924316, "rewards/repeat_soft/std": 0.020666038617491722, "rewards/judge_quality/mean": 0.38875001668930054, "rewards/judge_quality/std": 0.08675704896450043, "rewards/total_composite/mean": 0.7715330123901367, "rewards/total_composite/std": 0.03276064619421959, "reward": 0.7715330123901367, "reward_std": 0.03276064991950989, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10446494072675705, "sampling/sampling_logp_difference/max": 1.790860652923584, "sampling/importance_sampling_ratio/min": 0.16681653261184692, "sampling/importance_sampling_ratio/mean": 1.0059313774108887, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6116025559604168, "clip_ratio/low_mean": 0.04913809057325125, "clip_ratio/low_min": 0.04913809057325125, "clip_ratio/high_mean": 0.05452943313866854, "clip_ratio/high_max": 0.05452943313866854, "clip_ratio/region_mean": 0.10366752371191978, "reward_total_mean": 0.7715330123901367, "reward_meter_mean": 0.909159779548645, "reward_meter_std": 0.07416226714849472, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9578614234924316, "reward_repeat_soft_std": 0.020666038617491722, "reward_judge_quality_mean": 0.38875001668930054, "reward_judge_quality_std": 0.08675704896450043, "reward_total_composite_mean": 0.7715330123901367, "reward_total_composite_std": 0.03276064619421959} {"timestamp_utc": "2026-04-13T04:15:10Z", "mode": "train", "global_step": 2352, "epoch": 0.23626318432948268, "loss": -0.2005, "grad_norm": 1.3388198614120483, "learning_rate": 2.875757575757576e-06, "num_tokens": 4343215.0, "completions/mean_length": 158.0, "completions/min_length": 100.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 107.42857360839844, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.9026110172271729, "rewards/meter/std": 0.15281295776367188, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7354437112808228, "rewards/repeat_soft/std": 0.13316571712493896, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.6671110391616821, "rewards/total_composite/std": 0.27378836274147034, "reward": 0.6671110391616821, "reward_std": 0.27378836274147034, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06254018098115921, "sampling/sampling_logp_difference/max": 2.124619483947754, "sampling/importance_sampling_ratio/min": 0.11947842687368393, "sampling/importance_sampling_ratio/mean": 0.9957759380340576, "sampling/importance_sampling_ratio/max": 1.8450084924697876, "entropy": 0.28084208257496357, "clip_ratio/low_mean": 0.004347825888544321, "clip_ratio/low_min": 0.004347825888544321, "clip_ratio/high_mean": 0.04072360182181001, "clip_ratio/high_max": 0.04072360182181001, "clip_ratio/region_mean": 0.04507142771035433, "reward_total_mean": 0.6671110391616821, "reward_meter_mean": 0.9026110172271729, "reward_meter_std": 0.15281295776367188, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7354437112808228, "reward_repeat_soft_std": 0.13316571712493896, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.6671110391616821, "reward_total_composite_std": 0.27378836274147034} {"timestamp_utc": "2026-04-13T04:15:29Z", "mode": "train", "global_step": 2353, "epoch": 0.23636363636363636, "loss": -0.193, "grad_norm": 2.683708906173706, "learning_rate": 2.872727272727273e-06, "num_tokens": 4345234.0, "completions/mean_length": 151.375, "completions/min_length": 81.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 99.85714721679688, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.4529646039009094, "rewards/meter/std": 0.3486137390136719, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9785687923431396, "rewards/repeat_soft/std": 0.011535976082086563, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.28198277950286865, "rewards/total_composite/mean": 0.54149329662323, "rewards/total_composite/std": 0.2359592765569687, "reward": 0.54149329662323, "reward_std": 0.2359592765569687, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12961094081401825, "sampling/sampling_logp_difference/max": 3.028759241104126, "sampling/importance_sampling_ratio/min": 0.04837562143802643, "sampling/importance_sampling_ratio/mean": 1.0059984922409058, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5758643262088299, "clip_ratio/low_mean": 0.024425286799669266, "clip_ratio/low_min": 0.024425286799669266, "clip_ratio/high_mean": 0.11377029214054346, "clip_ratio/high_max": 0.11377029214054346, "clip_ratio/region_mean": 0.13819557894021273, "reward_total_mean": 0.54149329662323, "reward_meter_mean": 0.4529646039009094, "reward_meter_std": 0.3486137390136719, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9785687923431396, "reward_repeat_soft_std": 0.011535976082086563, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.28198277950286865, "reward_total_composite_mean": 0.54149329662323, "reward_total_composite_std": 0.2359592765569687} {"timestamp_utc": "2026-04-13T04:15:36Z", "mode": "train", "global_step": 2354, "epoch": 0.23646408839779007, "loss": 0.0572, "grad_norm": 13.033882141113281, "learning_rate": 2.86969696969697e-06, "num_tokens": 4346757.0, "completions/mean_length": 40.375, "completions/min_length": 35.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.375, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.8874404430389404, "rewards/meter/std": 0.20049948990345, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9824521541595459, "rewards/repeat_soft/std": 0.019077589735388756, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7747184038162231, "rewards/total_composite/std": 0.09078114479780197, "reward": 0.7747184038162231, "reward_std": 0.09078115969896317, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10938307642936707, "sampling/sampling_logp_difference/max": 1.8778727054595947, "sampling/importance_sampling_ratio/min": 0.15291506052017212, "sampling/importance_sampling_ratio/mean": 0.9978928565979004, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5304682590067387, "clip_ratio/low_mean": 0.011363636702299118, "clip_ratio/low_min": 0.011363636702299118, "clip_ratio/high_mean": 0.09972576377913356, "clip_ratio/high_max": 0.09972576377913356, "clip_ratio/region_mean": 0.11108940048143268, "reward_total_mean": 0.7747184038162231, "reward_meter_mean": 0.8874404430389404, "reward_meter_std": 0.20049948990345, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9824521541595459, "reward_repeat_soft_std": 0.019077589735388756, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7747184038162231, "reward_total_composite_std": 0.09078114479780197} {"timestamp_utc": "2026-04-13T04:15:44Z", "mode": "train", "global_step": 2355, "epoch": 0.23656454043194375, "loss": 0.0226, "grad_norm": 5.370815753936768, "learning_rate": 2.866666666666667e-06, "num_tokens": 4349267.0, "completions/mean_length": 146.75, "completions/min_length": 130.0, "completions/max_length": 161.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 146.75, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 161.0, "rewards/meter/mean": 0.8622211217880249, "rewards/meter/std": 0.21431687474250793, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8213942050933838, "rewards/repeat_soft/std": 0.07237093150615692, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.7206389307975769, "rewards/total_composite/std": 0.09471726417541504, "reward": 0.7206389307975769, "reward_std": 0.09471727162599564, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09270074963569641, "sampling/sampling_logp_difference/max": 1.9678833484649658, "sampling/importance_sampling_ratio/min": 0.1397523432970047, "sampling/importance_sampling_ratio/mean": 1.000538945198059, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5798838101327419, "clip_ratio/low_mean": 0.01923963101580739, "clip_ratio/low_min": 0.01923963101580739, "clip_ratio/high_mean": 0.06041077943518758, "clip_ratio/high_max": 0.06041077943518758, "clip_ratio/region_mean": 0.07965041045099497, "reward_total_mean": 0.7206389307975769, "reward_meter_mean": 0.8622211217880249, "reward_meter_std": 0.21431687474250793, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8213942050933838, "reward_repeat_soft_std": 0.07237093150615692, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.7206389307975769, "reward_total_composite_std": 0.09471726417541504} {"timestamp_utc": "2026-04-13T04:15:58Z", "mode": "train", "global_step": 2356, "epoch": 0.23666499246609743, "loss": -0.1114, "grad_norm": 1.4801408052444458, "learning_rate": 2.863636363636364e-06, "num_tokens": 4350794.0, "completions/mean_length": 162.875, "completions/min_length": 42.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 46.5, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.7745431065559387, "rewards/meter/std": 0.2655301094055176, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9925960898399353, "rewards/repeat_soft/std": 0.011051805689930916, "rewards/judge_quality/mean": 0.3462499976158142, "rewards/judge_quality/std": 0.18314221501350403, "rewards/total_composite/mean": 0.585649847984314, "rewards/total_composite/std": 0.3625388443470001, "reward": 0.585649847984314, "reward_std": 0.3625388443470001, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10265666991472244, "sampling/sampling_logp_difference/max": 1.0679802894592285, "sampling/importance_sampling_ratio/min": 0.3437019884586334, "sampling/importance_sampling_ratio/mean": 1.019695520401001, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5078098066151142, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.07793911872431636, "clip_ratio/high_max": 0.07793911872431636, "clip_ratio/region_mean": 0.07793911872431636, "reward_total_mean": 0.585649847984314, "reward_meter_mean": 0.7745431065559387, "reward_meter_std": 0.2655301094055176, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9925960898399353, "reward_repeat_soft_std": 0.011051805689930916, "reward_judge_quality_mean": 0.3462499976158142, "reward_judge_quality_std": 0.18314221501350403, "reward_total_composite_mean": 0.585649847984314, "reward_total_composite_std": 0.3625388443470001} {"timestamp_utc": "2026-04-13T04:16:10Z", "mode": "train", "global_step": 2357, "epoch": 0.23676544450025114, "loss": -0.1277, "grad_norm": 1.4434689283370972, "learning_rate": 2.860606060606061e-06, "num_tokens": 4352302.0, "completions/mean_length": 168.5, "completions/min_length": 51.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.8123019337654114, "rewards/meter/std": 0.31550338864326477, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.97359699010849, "rewards/repeat_soft/std": 0.024906961247324944, "rewards/judge_quality/mean": 0.3462499976158142, "rewards/judge_quality/std": 0.18314221501350403, "rewards/total_composite/mean": 0.6016125679016113, "rewards/total_composite/std": 0.37305334210395813, "reward": 0.6016125679016113, "reward_std": 0.37305334210395813, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09170068800449371, "sampling/sampling_logp_difference/max": 1.123824119567871, "sampling/importance_sampling_ratio/min": 0.3250344395637512, "sampling/importance_sampling_ratio/mean": 1.00330650806427, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39949873834848404, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0743113704957068, "clip_ratio/high_max": 0.0743113704957068, "clip_ratio/region_mean": 0.0743113704957068, "reward_total_mean": 0.6016125679016113, "reward_meter_mean": 0.8123019337654114, "reward_meter_std": 0.31550338864326477, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.97359699010849, "reward_repeat_soft_std": 0.024906961247324944, "reward_judge_quality_mean": 0.3462499976158142, "reward_judge_quality_std": 0.18314221501350403, "reward_total_composite_mean": 0.6016125679016113, "reward_total_composite_std": 0.37305334210395813} {"timestamp_utc": "2026-04-13T04:16:22Z", "mode": "train", "global_step": 2358, "epoch": 0.23686589653440482, "loss": -0.1833, "grad_norm": 2.108764886856079, "learning_rate": 2.857575757575758e-06, "num_tokens": 4354235.0, "completions/mean_length": 144.625, "completions/min_length": 85.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 92.14286041259766, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.7320746183395386, "rewards/meter/std": 0.33627864718437195, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9669666290283203, "rewards/repeat_soft/std": 0.028574280440807343, "rewards/judge_quality/mean": 0.45249998569488525, "rewards/judge_quality/std": 0.21625052392482758, "rewards/total_composite/mean": 0.6787552237510681, "rewards/total_composite/std": 0.28135019540786743, "reward": 0.6787552237510681, "reward_std": 0.28135019540786743, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09036403894424438, "sampling/sampling_logp_difference/max": 2.329291820526123, "sampling/importance_sampling_ratio/min": 0.09736467897891998, "sampling/importance_sampling_ratio/mean": 1.0061081647872925, "sampling/importance_sampling_ratio/max": 1.9519892930984497, "entropy": 0.3771265782415867, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.060894392314366996, "clip_ratio/high_max": 0.060894392314366996, "clip_ratio/region_mean": 0.060894392314366996, "reward_total_mean": 0.6787552237510681, "reward_meter_mean": 0.7320746183395386, "reward_meter_std": 0.33627864718437195, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9669666290283203, "reward_repeat_soft_std": 0.028574280440807343, "reward_judge_quality_mean": 0.45249998569488525, "reward_judge_quality_std": 0.21625052392482758, "reward_total_composite_mean": 0.6787552237510681, "reward_total_composite_std": 0.28135019540786743} {"timestamp_utc": "2026-04-13T04:16:31Z", "mode": "train", "global_step": 2359, "epoch": 0.23696634856855853, "loss": 0.6724, "grad_norm": 10.252663612365723, "learning_rate": 2.8545454545454548e-06, "num_tokens": 4356117.0, "completions/mean_length": 74.25, "completions/min_length": 46.0, "completions/max_length": 235.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.25, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 235.0, "rewards/meter/mean": 0.6690865755081177, "rewards/meter/std": 0.3774340748786926, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9657455682754517, "rewards/repeat_soft/std": 0.02534645050764084, "rewards/judge_quality/mean": 0.7487499713897705, "rewards/judge_quality/std": 0.332154780626297, "rewards/total_composite/mean": 0.7629134654998779, "rewards/total_composite/std": 0.28326231241226196, "reward": 0.7629134654998779, "reward_std": 0.2832622528076172, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10485774278640747, "sampling/sampling_logp_difference/max": 1.3118619918823242, "sampling/importance_sampling_ratio/min": 0.2693181037902832, "sampling/importance_sampling_ratio/mean": 1.0212129354476929, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5963715016841888, "clip_ratio/low_mean": 0.02306754607707262, "clip_ratio/low_min": 0.02306754607707262, "clip_ratio/high_mean": 0.056462487205863, "clip_ratio/high_max": 0.056462487205863, "clip_ratio/region_mean": 0.07953003328293562, "reward_total_mean": 0.7629134654998779, "reward_meter_mean": 0.6690865755081177, "reward_meter_std": 0.3774340748786926, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9657455682754517, "reward_repeat_soft_std": 0.02534645050764084, "reward_judge_quality_mean": 0.7487499713897705, "reward_judge_quality_std": 0.332154780626297, "reward_total_composite_mean": 0.7629134654998779, "reward_total_composite_std": 0.28326231241226196} {"timestamp_utc": "2026-04-13T04:16:38Z", "mode": "train", "global_step": 2360, "epoch": 0.2370668006027122, "loss": 0.009, "grad_norm": 4.854928970336914, "learning_rate": 2.8515151515151516e-06, "num_tokens": 4358381.0, "completions/mean_length": 114.0, "completions/min_length": 105.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.0, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.9787989258766174, "rewards/meter/std": 0.020095951855182648, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6475043296813965, "rewards/repeat_soft/std": 0.10725203901529312, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.7647099494934082, "rewards/total_composite/std": 0.046298544853925705, "reward": 0.7647099494934082, "reward_std": 0.0462985560297966, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06616435945034027, "sampling/sampling_logp_difference/max": 1.3656949996948242, "sampling/importance_sampling_ratio/min": 0.2552032470703125, "sampling/importance_sampling_ratio/mean": 0.9962505102157593, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2865325137972832, "clip_ratio/low_mean": 0.008753366768360138, "clip_ratio/low_min": 0.008753366768360138, "clip_ratio/high_mean": 0.04999238392338157, "clip_ratio/high_max": 0.04999238392338157, "clip_ratio/region_mean": 0.058745750691741705, "reward_total_mean": 0.7647099494934082, "reward_meter_mean": 0.9787989258766174, "reward_meter_std": 0.020095951855182648, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6475043296813965, "reward_repeat_soft_std": 0.10725203901529312, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.7647099494934082, "reward_total_composite_std": 0.046298544853925705} {"timestamp_utc": "2026-04-13T04:16:45Z", "mode": "train", "global_step": 2361, "epoch": 0.2371672526368659, "loss": 0.0015, "grad_norm": 7.515482425689697, "learning_rate": 2.848484848484849e-06, "num_tokens": 4360124.0, "completions/mean_length": 58.875, "completions/min_length": 52.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.875, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9798526763916016, "rewards/meter/std": 0.012577562592923641, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9431606531143188, "rewards/repeat_soft/std": 0.06251121312379837, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.23133157193660736, "rewards/total_composite/mean": 0.8322497606277466, "rewards/total_composite/std": 0.07522227615118027, "reward": 0.8322497606277466, "reward_std": 0.07522227615118027, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08342885971069336, "sampling/sampling_logp_difference/max": 1.1975879669189453, "sampling/importance_sampling_ratio/min": 0.3019215762615204, "sampling/importance_sampling_ratio/mean": 1.0015153884887695, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4853123798966408, "clip_ratio/low_mean": 0.05895052501000464, "clip_ratio/low_min": 0.05895052501000464, "clip_ratio/high_mean": 0.028629031963646412, "clip_ratio/high_max": 0.028629031963646412, "clip_ratio/region_mean": 0.08757955697365105, "reward_total_mean": 0.8322497606277466, "reward_meter_mean": 0.9798526763916016, "reward_meter_std": 0.012577562592923641, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9431606531143188, "reward_repeat_soft_std": 0.06251121312379837, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.23133157193660736, "reward_total_composite_mean": 0.8322497606277466, "reward_total_composite_std": 0.07522227615118027} {"timestamp_utc": "2026-04-13T04:16:53Z", "mode": "train", "global_step": 2362, "epoch": 0.2372677046710196, "loss": -0.0039, "grad_norm": 5.306748390197754, "learning_rate": 2.8454545454545457e-06, "num_tokens": 4362684.0, "completions/mean_length": 136.0, "completions/min_length": 124.0, "completions/max_length": 153.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.0, "completions/min_terminated_length": 124.0, "completions/max_terminated_length": 153.0, "rewards/meter/mean": 0.9798339605331421, "rewards/meter/std": 0.010603607632219791, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8128452301025391, "rewards/repeat_soft/std": 0.0536111518740654, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7982098460197449, "rewards/total_composite/std": 0.00613095099106431, "reward": 0.7982098460197449, "reward_std": 0.0061309547163546085, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07715856283903122, "sampling/sampling_logp_difference/max": 5.030965805053711, "sampling/importance_sampling_ratio/min": 0.0065324981696903706, "sampling/importance_sampling_ratio/mean": 1.008824110031128, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4240390695631504, "clip_ratio/low_mean": 0.041710769291967154, "clip_ratio/low_min": 0.041710769291967154, "clip_ratio/high_mean": 0.02016149740666151, "clip_ratio/high_max": 0.02016149740666151, "clip_ratio/region_mean": 0.061872266698628664, "reward_total_mean": 0.7982098460197449, "reward_meter_mean": 0.9798339605331421, "reward_meter_std": 0.010603607632219791, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8128452301025391, "reward_repeat_soft_std": 0.0536111518740654, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7982098460197449, "reward_total_composite_std": 0.00613095099106431} {"timestamp_utc": "2026-04-13T04:17:04Z", "mode": "train", "global_step": 2363, "epoch": 0.23736815670517328, "loss": -0.2037, "grad_norm": 1.6761952638626099, "learning_rate": 2.8424242424242425e-06, "num_tokens": 4364958.0, "completions/mean_length": 156.25, "completions/min_length": 95.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 105.42857360839844, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.8684190511703491, "rewards/meter/std": 0.35081154108047485, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.96302729845047, "rewards/repeat_soft/std": 0.031363293528556824, "rewards/judge_quality/mean": 0.3399999737739563, "rewards/judge_quality/std": 0.1505228877067566, "rewards/total_composite/mean": 0.7064194679260254, "rewards/total_composite/std": 0.2867966592311859, "reward": 0.7064194679260254, "reward_std": 0.2867966592311859, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12172281742095947, "sampling/sampling_logp_difference/max": 1.9301800727844238, "sampling/importance_sampling_ratio/min": 0.14512206614017487, "sampling/importance_sampling_ratio/mean": 1.0038666725158691, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6735124066472054, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10654845554381609, "clip_ratio/high_max": 0.10654845554381609, "clip_ratio/region_mean": 0.10654845554381609, "reward_total_mean": 0.7064194679260254, "reward_meter_mean": 0.8684190511703491, "reward_meter_std": 0.35081154108047485, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.96302729845047, "reward_repeat_soft_std": 0.031363293528556824, "reward_judge_quality_mean": 0.3399999737739563, "reward_judge_quality_std": 0.1505228877067566, "reward_total_composite_mean": 0.7064194679260254, "reward_total_composite_std": 0.2867966592311859} {"timestamp_utc": "2026-04-13T04:17:12Z", "mode": "train", "global_step": 2364, "epoch": 0.23746860873932696, "loss": 0.022, "grad_norm": 6.185801029205322, "learning_rate": 2.83939393939394e-06, "num_tokens": 4367747.0, "completions/mean_length": 145.625, "completions/min_length": 135.0, "completions/max_length": 154.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 145.625, "completions/min_terminated_length": 135.0, "completions/max_terminated_length": 154.0, "rewards/meter/mean": 0.9845558404922485, "rewards/meter/std": 0.0225363876670599, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8739386796951294, "rewards/repeat_soft/std": 0.06137301027774811, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.7835689783096313, "rewards/total_composite/std": 0.039025284349918365, "reward": 0.7835689783096313, "reward_std": 0.039025284349918365, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08937781304121017, "sampling/sampling_logp_difference/max": 3.9322903156280518, "sampling/importance_sampling_ratio/min": 0.019598733633756638, "sampling/importance_sampling_ratio/mean": 1.0061582326889038, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4800117537379265, "clip_ratio/low_mean": 0.028110009618103504, "clip_ratio/low_min": 0.028110009618103504, "clip_ratio/high_mean": 0.0578347435221076, "clip_ratio/high_max": 0.0578347435221076, "clip_ratio/region_mean": 0.0859447531402111, "reward_total_mean": 0.7835689783096313, "reward_meter_mean": 0.9845558404922485, "reward_meter_std": 0.0225363876670599, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8739386796951294, "reward_repeat_soft_std": 0.06137301027774811, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.7835689783096313, "reward_total_composite_std": 0.039025284349918365} {"timestamp_utc": "2026-04-13T04:17:19Z", "mode": "train", "global_step": 2365, "epoch": 0.23756906077348067, "loss": 0.0503, "grad_norm": 12.16795825958252, "learning_rate": 2.8363636363636366e-06, "num_tokens": 4369573.0, "completions/mean_length": 65.25, "completions/min_length": 59.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.25, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.8007506132125854, "rewards/meter/std": 0.18110552430152893, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9720698595046997, "rewards/repeat_soft/std": 0.009579234756529331, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.7460447549819946, "rewards/total_composite/std": 0.11300327628850937, "reward": 0.7460447549819946, "reward_std": 0.11300326138734818, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10633543133735657, "sampling/sampling_logp_difference/max": 1.6447038650512695, "sampling/importance_sampling_ratio/min": 0.193069726228714, "sampling/importance_sampling_ratio/mean": 1.0085797309875488, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49294867366552353, "clip_ratio/low_mean": 0.030562865547835827, "clip_ratio/low_min": 0.030562865547835827, "clip_ratio/high_mean": 0.05457504512742162, "clip_ratio/high_max": 0.05457504512742162, "clip_ratio/region_mean": 0.08513791067525744, "reward_total_mean": 0.7460447549819946, "reward_meter_mean": 0.8007506132125854, "reward_meter_std": 0.18110552430152893, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9720698595046997, "reward_repeat_soft_std": 0.009579234756529331, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.7460447549819946, "reward_total_composite_std": 0.11300327628850937} {"timestamp_utc": "2026-04-13T04:17:27Z", "mode": "train", "global_step": 2366, "epoch": 0.23766951280763435, "loss": -0.0165, "grad_norm": 8.84438419342041, "learning_rate": 2.8333333333333335e-06, "num_tokens": 4371444.0, "completions/mean_length": 60.875, "completions/min_length": 54.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9922090172767639, "rewards/meter/std": 0.0015462747542187572, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9412178993225098, "rewards/repeat_soft/std": 0.025463717058300972, "rewards/judge_quality/mean": 0.32499998807907104, "rewards/judge_quality/std": 0.15212775766849518, "rewards/total_composite/mean": 0.7881158590316772, "rewards/total_composite/std": 0.04484758898615837, "reward": 0.7881158590316772, "reward_std": 0.04484757035970688, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08907467126846313, "sampling/sampling_logp_difference/max": 1.508345603942871, "sampling/importance_sampling_ratio/min": 0.22127576172351837, "sampling/importance_sampling_ratio/mean": 1.0036734342575073, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5111229494214058, "clip_ratio/low_mean": 0.019428979139775038, "clip_ratio/low_min": 0.019428979139775038, "clip_ratio/high_mean": 0.06132835615426302, "clip_ratio/high_max": 0.06132835615426302, "clip_ratio/region_mean": 0.08075733529403806, "reward_total_mean": 0.7881158590316772, "reward_meter_mean": 0.9922090172767639, "reward_meter_std": 0.0015462747542187572, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9412178993225098, "reward_repeat_soft_std": 0.025463717058300972, "reward_judge_quality_mean": 0.32499998807907104, "reward_judge_quality_std": 0.15212775766849518, "reward_total_composite_mean": 0.7881158590316772, "reward_total_composite_std": 0.04484758898615837} {"timestamp_utc": "2026-04-13T04:17:34Z", "mode": "train", "global_step": 2367, "epoch": 0.23776996484178806, "loss": 0.0668, "grad_norm": 10.209254264831543, "learning_rate": 2.8303030303030303e-06, "num_tokens": 4373272.0, "completions/mean_length": 61.5, "completions/min_length": 53.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.5, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9720644950866699, "rewards/meter/std": 0.021038595587015152, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9619510769844055, "rewards/repeat_soft/std": 0.031078077852725983, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.19078318774700165, "rewards/total_composite/mean": 0.8407491445541382, "rewards/total_composite/std": 0.0634908452630043, "reward": 0.8407491445541382, "reward_std": 0.0634908527135849, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11312397569417953, "sampling/sampling_logp_difference/max": 1.5663433074951172, "sampling/importance_sampling_ratio/min": 0.2088073343038559, "sampling/importance_sampling_ratio/mean": 1.0037519931793213, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4848780333995819, "clip_ratio/low_mean": 0.07016055379062891, "clip_ratio/low_min": 0.07016055379062891, "clip_ratio/high_mean": 0.018829884007573128, "clip_ratio/high_max": 0.018829884007573128, "clip_ratio/region_mean": 0.08899043779820204, "reward_total_mean": 0.8407491445541382, "reward_meter_mean": 0.9720644950866699, "reward_meter_std": 0.021038595587015152, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9619510769844055, "reward_repeat_soft_std": 0.031078077852725983, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.19078318774700165, "reward_total_composite_mean": 0.8407491445541382, "reward_total_composite_std": 0.0634908452630043} {"timestamp_utc": "2026-04-13T04:17:42Z", "mode": "train", "global_step": 2368, "epoch": 0.23787041687594174, "loss": 0.058, "grad_norm": 5.401748180389404, "learning_rate": 2.8272727272727275e-06, "num_tokens": 4376049.0, "completions/mean_length": 165.125, "completions/min_length": 154.0, "completions/max_length": 181.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 165.125, "completions/min_terminated_length": 154.0, "completions/max_terminated_length": 181.0, "rewards/meter/mean": 0.9818468689918518, "rewards/meter/std": 0.027699744328856468, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8940421342849731, "rewards/repeat_soft/std": 0.07651467621326447, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7881103157997131, "rewards/total_composite/std": 0.028832867741584778, "reward": 0.7881103157997131, "reward_std": 0.028832871466875076, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0924229770898819, "sampling/sampling_logp_difference/max": 1.5387392044067383, "sampling/importance_sampling_ratio/min": 0.21465158462524414, "sampling/importance_sampling_ratio/mean": 1.0064382553100586, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5605798363685608, "clip_ratio/low_mean": 0.0336351552978158, "clip_ratio/low_min": 0.0336351552978158, "clip_ratio/high_mean": 0.056401670910418034, "clip_ratio/high_max": 0.056401670910418034, "clip_ratio/region_mean": 0.09003682620823383, "reward_total_mean": 0.7881103157997131, "reward_meter_mean": 0.9818468689918518, "reward_meter_std": 0.027699744328856468, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8940421342849731, "reward_repeat_soft_std": 0.07651467621326447, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7881103157997131, "reward_total_composite_std": 0.028832867741584778} {"timestamp_utc": "2026-04-13T04:17:48Z", "mode": "train", "global_step": 2369, "epoch": 0.23797086891009542, "loss": 0.0215, "grad_norm": 15.925398826599121, "learning_rate": 2.8242424242424244e-06, "num_tokens": 4377610.0, "completions/mean_length": 26.125, "completions/min_length": 20.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.125, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.9542003870010376, "rewards/meter/std": 0.09192883223295212, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9477933049201965, "rewards/repeat_soft/std": 0.025734715163707733, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8035445213317871, "rewards/total_composite/std": 0.03908426687121391, "reward": 0.8035445213317871, "reward_std": 0.03908428177237511, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1090904250741005, "sampling/sampling_logp_difference/max": 1.4815702438354492, "sampling/importance_sampling_ratio/min": 0.22728052735328674, "sampling/importance_sampling_ratio/mean": 1.011565089225769, "sampling/importance_sampling_ratio/max": 1.6049357652664185, "entropy": 0.658360555768013, "clip_ratio/low_mean": 0.009615384973585606, "clip_ratio/low_min": 0.009615384973585606, "clip_ratio/high_mean": 0.0974651095457375, "clip_ratio/high_max": 0.0974651095457375, "clip_ratio/region_mean": 0.10708049451932311, "reward_total_mean": 0.8035445213317871, "reward_meter_mean": 0.9542003870010376, "reward_meter_std": 0.09192883223295212, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9477933049201965, "reward_repeat_soft_std": 0.025734715163707733, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8035445213317871, "reward_total_composite_std": 0.03908426687121391} {"timestamp_utc": "2026-04-13T04:17:59Z", "mode": "train", "global_step": 2370, "epoch": 0.23807132094424913, "loss": -0.0862, "grad_norm": 2.447359561920166, "learning_rate": 2.821212121212121e-06, "num_tokens": 4378800.0, "completions/mean_length": 89.75, "completions/min_length": 25.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 29.428573608398438, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.7071264386177063, "rewards/meter/std": 0.29919418692588806, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9593632221221924, "rewards/repeat_soft/std": 0.006439989898353815, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.15315140783786774, "rewards/total_composite/mean": 0.6029156446456909, "rewards/total_composite/std": 0.27854180335998535, "reward": 0.6029156446456909, "reward_std": 0.27854177355766296, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12088526785373688, "sampling/sampling_logp_difference/max": 1.0554585456848145, "sampling/importance_sampling_ratio/min": 0.34803280234336853, "sampling/importance_sampling_ratio/mean": 1.0107276439666748, "sampling/importance_sampling_ratio/max": 1.8466118574142456, "entropy": 0.98019989579916, "clip_ratio/low_mean": 0.02636363636702299, "clip_ratio/low_min": 0.02636363636702299, "clip_ratio/high_mean": 0.0917050689458847, "clip_ratio/high_max": 0.0917050689458847, "clip_ratio/region_mean": 0.1180687053129077, "reward_total_mean": 0.6029156446456909, "reward_meter_mean": 0.7071264386177063, "reward_meter_std": 0.29919418692588806, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9593632221221924, "reward_repeat_soft_std": 0.006439989898353815, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.15315140783786774, "reward_total_composite_mean": 0.6029156446456909, "reward_total_composite_std": 0.27854180335998535} {"timestamp_utc": "2026-04-13T04:18:06Z", "mode": "train", "global_step": 2371, "epoch": 0.2381717729784028, "loss": 0.0395, "grad_norm": 5.902582168579102, "learning_rate": 2.818181818181818e-06, "num_tokens": 4381003.0, "completions/mean_length": 108.375, "completions/min_length": 100.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.375, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.9898571968078613, "rewards/meter/std": 0.005854758433997631, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7593591809272766, "rewards/repeat_soft/std": 0.06765656918287277, "rewards/judge_quality/mean": 0.30124998092651367, "rewards/judge_quality/std": 0.10398317128419876, "rewards/total_composite/mean": 0.7617466449737549, "rewards/total_composite/std": 0.03475917875766754, "reward": 0.7617466449737549, "reward_std": 0.03475918993353844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06337554007768631, "sampling/sampling_logp_difference/max": 1.295915126800537, "sampling/importance_sampling_ratio/min": 0.3847007155418396, "sampling/importance_sampling_ratio/mean": 1.0123465061187744, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.35449371859431267, "clip_ratio/low_mean": 0.031681555323302746, "clip_ratio/low_min": 0.031681555323302746, "clip_ratio/high_mean": 0.0191863258369267, "clip_ratio/high_max": 0.0191863258369267, "clip_ratio/region_mean": 0.050867881160229445, "reward_total_mean": 0.7617466449737549, "reward_meter_mean": 0.9898571968078613, "reward_meter_std": 0.005854758433997631, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7593591809272766, "reward_repeat_soft_std": 0.06765656918287277, "reward_judge_quality_mean": 0.30124998092651367, "reward_judge_quality_std": 0.10398317128419876, "reward_total_composite_mean": 0.7617466449737549, "reward_total_composite_std": 0.03475917875766754} {"timestamp_utc": "2026-04-13T04:18:14Z", "mode": "train", "global_step": 2372, "epoch": 0.23827222501255652, "loss": 0.0371, "grad_norm": 6.77298641204834, "learning_rate": 2.8151515151515153e-06, "num_tokens": 4383552.0, "completions/mean_length": 118.625, "completions/min_length": 108.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.625, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.9804316759109497, "rewards/meter/std": 0.007503341417759657, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9325956106185913, "rewards/repeat_soft/std": 0.02620098739862442, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8115788102149963, "rewards/total_composite/std": 0.0047071813605725765, "reward": 0.8115788102149963, "reward_std": 0.004707160405814648, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09232497960329056, "sampling/sampling_logp_difference/max": 1.8815083503723145, "sampling/importance_sampling_ratio/min": 0.2574116885662079, "sampling/importance_sampling_ratio/mean": 1.0042849779129028, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5545084066689014, "clip_ratio/low_mean": 0.044058968080207705, "clip_ratio/low_min": 0.044058968080207705, "clip_ratio/high_mean": 0.04001238755881786, "clip_ratio/high_max": 0.04001238755881786, "clip_ratio/region_mean": 0.08407135563902557, "reward_total_mean": 0.8115788102149963, "reward_meter_mean": 0.9804316759109497, "reward_meter_std": 0.007503341417759657, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9325956106185913, "reward_repeat_soft_std": 0.02620098739862442, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8115788102149963, "reward_total_composite_std": 0.0047071813605725765} {"timestamp_utc": "2026-04-13T04:18:25Z", "mode": "train", "global_step": 2373, "epoch": 0.2383726770467102, "loss": -0.0937, "grad_norm": 3.2704617977142334, "learning_rate": 2.812121212121212e-06, "num_tokens": 4385040.0, "completions/mean_length": 93.0, "completions/min_length": 30.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 33.142860412597656, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.6521397829055786, "rewards/meter/std": 0.4138062000274658, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9622748494148254, "rewards/repeat_soft/std": 0.0006367546156980097, "rewards/judge_quality/mean": 0.39625000953674316, "rewards/judge_quality/std": 0.14029940962791443, "rewards/total_composite/mean": 0.6245783567428589, "rewards/total_composite/std": 0.2917555272579193, "reward": 0.6245783567428589, "reward_std": 0.2917555272579193, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13073207437992096, "sampling/sampling_logp_difference/max": 1.12662672996521, "sampling/importance_sampling_ratio/min": 0.3241247832775116, "sampling/importance_sampling_ratio/mean": 1.011419415473938, "sampling/importance_sampling_ratio/max": 1.650099515914917, "entropy": 0.5585838072001934, "clip_ratio/low_mean": 0.043560607358813286, "clip_ratio/low_min": 0.043560607358813286, "clip_ratio/high_mean": 0.09235965274274349, "clip_ratio/high_max": 0.09235965274274349, "clip_ratio/region_mean": 0.13592026010155678, "reward_total_mean": 0.6245783567428589, "reward_meter_mean": 0.6521397829055786, "reward_meter_std": 0.4138062000274658, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9622748494148254, "reward_repeat_soft_std": 0.0006367546156980097, "reward_judge_quality_mean": 0.39625000953674316, "reward_judge_quality_std": 0.14029940962791443, "reward_total_composite_mean": 0.6245783567428589, "reward_total_composite_std": 0.2917555272579193} {"timestamp_utc": "2026-04-13T04:18:32Z", "mode": "train", "global_step": 2374, "epoch": 0.23847312908086388, "loss": 0.0195, "grad_norm": 10.221879005432129, "learning_rate": 2.809090909090909e-06, "num_tokens": 4386647.0, "completions/mean_length": 28.875, "completions/min_length": 26.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.875, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9807325601577759, "rewards/meter/std": 0.014163346029818058, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9448447823524475, "rewards/repeat_soft/std": 0.033727385103702545, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.8316891193389893, "rewards/total_composite/std": 0.05060151219367981, "reward": 0.8316891193389893, "reward_std": 0.050601497292518616, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09940163046121597, "sampling/sampling_logp_difference/max": 0.9557771682739258, "sampling/importance_sampling_ratio/min": 0.3845131993293762, "sampling/importance_sampling_ratio/mean": 1.0073561668395996, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6944334283471107, "clip_ratio/low_mean": 0.0596794025041163, "clip_ratio/low_min": 0.0596794025041163, "clip_ratio/high_mean": 0.018518518656492233, "clip_ratio/high_max": 0.018518518656492233, "clip_ratio/region_mean": 0.07819792116060853, "reward_total_mean": 0.8316891193389893, "reward_meter_mean": 0.9807325601577759, "reward_meter_std": 0.014163346029818058, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9448447823524475, "reward_repeat_soft_std": 0.033727385103702545, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.8316891193389893, "reward_total_composite_std": 0.05060151219367981} {"timestamp_utc": "2026-04-13T04:18:38Z", "mode": "train", "global_step": 2375, "epoch": 0.23857358111501759, "loss": 0.0539, "grad_norm": 13.6381254196167, "learning_rate": 2.806060606060606e-06, "num_tokens": 4388153.0, "completions/mean_length": 29.25, "completions/min_length": 25.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.25, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.8796474933624268, "rewards/meter/std": 0.26121780276298523, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9082473516464233, "rewards/repeat_soft/std": 0.08430705964565277, "rewards/judge_quality/mean": 0.39249998331069946, "rewards/judge_quality/std": 0.08892211318016052, "rewards/total_composite/mean": 0.7544161081314087, "rewards/total_composite/std": 0.118186816573143, "reward": 0.7544161081314087, "reward_std": 0.118186816573143, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10941938310861588, "sampling/sampling_logp_difference/max": 0.9987757205963135, "sampling/importance_sampling_ratio/min": 0.36833012104034424, "sampling/importance_sampling_ratio/mean": 1.0101399421691895, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6657619997859001, "clip_ratio/low_mean": 0.0297413794323802, "clip_ratio/low_min": 0.0297413794323802, "clip_ratio/high_mean": 0.07646737154573202, "clip_ratio/high_max": 0.07646737154573202, "clip_ratio/region_mean": 0.10620875097811222, "reward_total_mean": 0.7544161081314087, "reward_meter_mean": 0.8796474933624268, "reward_meter_std": 0.26121780276298523, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9082473516464233, "reward_repeat_soft_std": 0.08430705964565277, "reward_judge_quality_mean": 0.39249998331069946, "reward_judge_quality_std": 0.08892211318016052, "reward_total_composite_mean": 0.7544161081314087, "reward_total_composite_std": 0.118186816573143} {"timestamp_utc": "2026-04-13T04:18:46Z", "mode": "train", "global_step": 2376, "epoch": 0.23867403314917127, "loss": 0.0136, "grad_norm": 4.284181594848633, "learning_rate": 2.803030303030303e-06, "num_tokens": 4391055.0, "completions/mean_length": 175.75, "completions/min_length": 163.0, "completions/max_length": 191.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 175.75, "completions/min_terminated_length": 163.0, "completions/max_terminated_length": 191.0, "rewards/meter/mean": 0.9911484718322754, "rewards/meter/std": 0.005596070550382137, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.844290018081665, "rewards/repeat_soft/std": 0.07255405932664871, "rewards/judge_quality/mean": 0.2462499886751175, "rewards/judge_quality/std": 0.08348438143730164, "rewards/total_composite/mean": 0.7505707740783691, "rewards/total_composite/std": 0.03276975080370903, "reward": 0.7505707740783691, "reward_std": 0.03276974335312843, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0765308067202568, "sampling/sampling_logp_difference/max": 1.752817153930664, "sampling/importance_sampling_ratio/min": 0.1732850819826126, "sampling/importance_sampling_ratio/mean": 1.017872929573059, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4995374083518982, "clip_ratio/low_mean": 0.021753760520368814, "clip_ratio/low_min": 0.021753760520368814, "clip_ratio/high_mean": 0.042158594354987144, "clip_ratio/high_max": 0.042158594354987144, "clip_ratio/region_mean": 0.06391235487535596, "reward_total_mean": 0.7505707740783691, "reward_meter_mean": 0.9911484718322754, "reward_meter_std": 0.005596070550382137, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.844290018081665, "reward_repeat_soft_std": 0.07255405932664871, "reward_judge_quality_mean": 0.2462499886751175, "reward_judge_quality_std": 0.08348438143730164, "reward_total_composite_mean": 0.7505707740783691, "reward_total_composite_std": 0.03276975080370903} {"timestamp_utc": "2026-04-13T04:18:55Z", "mode": "train", "global_step": 2377, "epoch": 0.23877448518332497, "loss": 0.0057, "grad_norm": 3.4988186359405518, "learning_rate": 2.8000000000000003e-06, "num_tokens": 4394247.0, "completions/mean_length": 200.0, "completions/min_length": 178.0, "completions/max_length": 214.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 200.0, "completions/min_terminated_length": 178.0, "completions/max_terminated_length": 214.0, "rewards/meter/mean": 0.9801251292228699, "rewards/meter/std": 0.010416720993816853, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7606629133224487, "rewards/repeat_soft/std": 0.1004609763622284, "rewards/judge_quality/mean": 0.2799999713897705, "rewards/judge_quality/std": 0.09304375946521759, "rewards/total_composite/mean": 0.7511225938796997, "rewards/total_composite/std": 0.03377591073513031, "reward": 0.7511225938796997, "reward_std": 0.03377590700984001, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06039508804678917, "sampling/sampling_logp_difference/max": 2.4506759643554688, "sampling/importance_sampling_ratio/min": 0.08623527735471725, "sampling/importance_sampling_ratio/mean": 1.003098726272583, "sampling/importance_sampling_ratio/max": 1.994307279586792, "entropy": 0.3562541641294956, "clip_ratio/low_mean": 0.04102576000150293, "clip_ratio/low_min": 0.04102576000150293, "clip_ratio/high_mean": 0.02021271362900734, "clip_ratio/high_max": 0.02021271362900734, "clip_ratio/region_mean": 0.06123847363051027, "reward_total_mean": 0.7511225938796997, "reward_meter_mean": 0.9801251292228699, "reward_meter_std": 0.010416720993816853, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7606629133224487, "reward_repeat_soft_std": 0.1004609763622284, "reward_judge_quality_mean": 0.2799999713897705, "reward_judge_quality_std": 0.09304375946521759, "reward_total_composite_mean": 0.7511225938796997, "reward_total_composite_std": 0.03377591073513031} {"timestamp_utc": "2026-04-13T04:19:02Z", "mode": "train", "global_step": 2378, "epoch": 0.23887493721747866, "loss": 0.005, "grad_norm": 7.663424968719482, "learning_rate": 2.7969696969696976e-06, "num_tokens": 4396092.0, "completions/mean_length": 52.625, "completions/min_length": 46.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.625, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.988467812538147, "rewards/meter/std": 0.007351188454777002, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8703055381774902, "rewards/repeat_soft/std": 0.10983171314001083, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.8277161121368408, "rewards/total_composite/std": 0.0560237281024456, "reward": 0.8277161121368408, "reward_std": 0.0560237281024456, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07491254061460495, "sampling/sampling_logp_difference/max": 1.493638515472412, "sampling/importance_sampling_ratio/min": 0.2245541214942932, "sampling/importance_sampling_ratio/mean": 1.0115103721618652, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39774760603904724, "clip_ratio/low_mean": 0.05418275622650981, "clip_ratio/low_min": 0.05418275622650981, "clip_ratio/high_mean": 0.0072115384973585606, "clip_ratio/high_max": 0.0072115384973585606, "clip_ratio/region_mean": 0.06139429472386837, "reward_total_mean": 0.8277161121368408, "reward_meter_mean": 0.988467812538147, "reward_meter_std": 0.007351188454777002, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8703055381774902, "reward_repeat_soft_std": 0.10983171314001083, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.8277161121368408, "reward_total_composite_std": 0.0560237281024456} {"timestamp_utc": "2026-04-13T04:19:13Z", "mode": "train", "global_step": 2379, "epoch": 0.23897538925163234, "loss": -0.1276, "grad_norm": 1.8265175819396973, "learning_rate": 2.7939393939393944e-06, "num_tokens": 4397633.0, "completions/mean_length": 104.625, "completions/min_length": 36.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 46.42857360839844, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9640837907791138, "rewards/meter/std": 0.04575441777706146, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9656071662902832, "rewards/repeat_soft/std": 0.03340619057416916, "rewards/judge_quality/mean": 0.6025000214576721, "rewards/judge_quality/std": 0.24766048789024353, "rewards/total_composite/mean": 0.780593752861023, "rewards/total_composite/std": 0.3171443045139313, "reward": 0.780593752861023, "reward_std": 0.3171443045139313, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12157497555017471, "sampling/sampling_logp_difference/max": 1.4377946853637695, "sampling/importance_sampling_ratio/min": 0.23745083808898926, "sampling/importance_sampling_ratio/mean": 0.9980635046958923, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48234083876013756, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08203817997127771, "clip_ratio/high_max": 0.08203817997127771, "clip_ratio/region_mean": 0.08203817997127771, "reward_total_mean": 0.780593752861023, "reward_meter_mean": 0.9640837907791138, "reward_meter_std": 0.04575441777706146, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9656071662902832, "reward_repeat_soft_std": 0.03340619057416916, "reward_judge_quality_mean": 0.6025000214576721, "reward_judge_quality_std": 0.24766048789024353, "reward_total_composite_mean": 0.780593752861023, "reward_total_composite_std": 0.3171443045139313} {"timestamp_utc": "2026-04-13T04:19:20Z", "mode": "train", "global_step": 2380, "epoch": 0.23907584128578604, "loss": 0.0256, "grad_norm": 8.870682716369629, "learning_rate": 2.7909090909090912e-06, "num_tokens": 4399338.0, "completions/mean_length": 61.125, "completions/min_length": 54.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.125, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9907095432281494, "rewards/meter/std": 0.005707127042114735, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.966938853263855, "rewards/repeat_soft/std": 0.042921263724565506, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.8560131788253784, "rewards/total_composite/std": 0.06834213435649872, "reward": 0.8560131788253784, "reward_std": 0.06834214180707932, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1068754568696022, "sampling/sampling_logp_difference/max": 1.7470126152038574, "sampling/importance_sampling_ratio/min": 0.17429384589195251, "sampling/importance_sampling_ratio/mean": 0.996345579624176, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6068036817014217, "clip_ratio/low_mean": 0.059246668592095375, "clip_ratio/low_min": 0.059246668592095375, "clip_ratio/high_mean": 0.028118192218244076, "clip_ratio/high_max": 0.028118192218244076, "clip_ratio/region_mean": 0.08736486081033945, "reward_total_mean": 0.8560131788253784, "reward_meter_mean": 0.9907095432281494, "reward_meter_std": 0.005707127042114735, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.966938853263855, "reward_repeat_soft_std": 0.042921263724565506, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.8560131788253784, "reward_total_composite_std": 0.06834213435649872} {"timestamp_utc": "2026-04-13T04:19:28Z", "mode": "train", "global_step": 2381, "epoch": 0.23917629331993973, "loss": 0.015, "grad_norm": 6.359300136566162, "learning_rate": 2.7878787878787885e-06, "num_tokens": 4401669.0, "completions/mean_length": 114.375, "completions/min_length": 110.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.375, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.8938387632369995, "rewards/meter/std": 0.17841114103794098, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9106141328811646, "rewards/repeat_soft/std": 0.052521560341119766, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.1524970829486847, "rewards/total_composite/mean": 0.770413875579834, "rewards/total_composite/std": 0.0947524830698967, "reward": 0.770413875579834, "reward_std": 0.0947524681687355, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08397268503904343, "sampling/sampling_logp_difference/max": 1.9394028186798096, "sampling/importance_sampling_ratio/min": 0.14378979802131653, "sampling/importance_sampling_ratio/mean": 1.0066413879394531, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4734833352267742, "clip_ratio/low_mean": 0.030097721144557, "clip_ratio/low_min": 0.030097721144557, "clip_ratio/high_mean": 0.05299512017518282, "clip_ratio/high_max": 0.05299512017518282, "clip_ratio/region_mean": 0.08309284131973982, "reward_total_mean": 0.770413875579834, "reward_meter_mean": 0.8938387632369995, "reward_meter_std": 0.17841114103794098, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9106141328811646, "reward_repeat_soft_std": 0.052521560341119766, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.1524970829486847, "reward_total_composite_mean": 0.770413875579834, "reward_total_composite_std": 0.0947524830698967} {"timestamp_utc": "2026-04-13T04:19:34Z", "mode": "train", "global_step": 2382, "epoch": 0.23927674535409343, "loss": 0.0808, "grad_norm": 14.636855125427246, "learning_rate": 2.7848484848484853e-06, "num_tokens": 4403241.0, "completions/mean_length": 34.5, "completions/min_length": 29.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.5, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.9136193990707397, "rewards/meter/std": 0.17172162234783173, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9585226774215698, "rewards/repeat_soft/std": 0.011249415576457977, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7886060476303101, "rewards/total_composite/std": 0.07601718604564667, "reward": 0.7886060476303101, "reward_std": 0.07601717859506607, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12978307902812958, "sampling/sampling_logp_difference/max": 3.220818281173706, "sampling/importance_sampling_ratio/min": 0.03992237523198128, "sampling/importance_sampling_ratio/mean": 1.0053999423980713, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6957956403493881, "clip_ratio/low_mean": 0.015625, "clip_ratio/low_min": 0.015625, "clip_ratio/high_mean": 0.13227378716692328, "clip_ratio/high_max": 0.13227378716692328, "clip_ratio/region_mean": 0.14789878716692328, "reward_total_mean": 0.7886060476303101, "reward_meter_mean": 0.9136193990707397, "reward_meter_std": 0.17172162234783173, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9585226774215698, "reward_repeat_soft_std": 0.011249415576457977, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7886060476303101, "reward_total_composite_std": 0.07601718604564667} {"timestamp_utc": "2026-04-13T04:19:41Z", "mode": "train", "global_step": 2383, "epoch": 0.23937719738824711, "loss": -0.0065, "grad_norm": 7.145590305328369, "learning_rate": 2.781818181818182e-06, "num_tokens": 4405532.0, "completions/mean_length": 97.375, "completions/min_length": 91.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.375, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.9839315414428711, "rewards/meter/std": 0.014994838275015354, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9305007457733154, "rewards/repeat_soft/std": 0.05491577461361885, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8118192553520203, "rewards/total_composite/std": 0.005402666982263327, "reward": 0.8118192553520203, "reward_std": 0.005402679089456797, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08531832695007324, "sampling/sampling_logp_difference/max": 1.6879031658172607, "sampling/importance_sampling_ratio/min": 0.18490682542324066, "sampling/importance_sampling_ratio/mean": 1.0152571201324463, "sampling/importance_sampling_ratio/max": 1.7224751710891724, "entropy": 0.5671970807015896, "clip_ratio/low_mean": 0.04355962201952934, "clip_ratio/low_min": 0.04355962201952934, "clip_ratio/high_mean": 0.019819581881165504, "clip_ratio/high_max": 0.019819581881165504, "clip_ratio/region_mean": 0.06337920390069485, "reward_total_mean": 0.8118192553520203, "reward_meter_mean": 0.9839315414428711, "reward_meter_std": 0.014994838275015354, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9305007457733154, "reward_repeat_soft_std": 0.05491577461361885, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8118192553520203, "reward_total_composite_std": 0.005402666982263327} {"timestamp_utc": "2026-04-13T04:19:53Z", "mode": "train", "global_step": 2384, "epoch": 0.2394776494224008, "loss": -0.2347, "grad_norm": 1.3276218175888062, "learning_rate": 2.778787878787879e-06, "num_tokens": 4408178.0, "completions/mean_length": 221.75, "completions/min_length": 155.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 180.2857208251953, "completions/min_terminated_length": 155.0, "completions/max_terminated_length": 211.0, "rewards/meter/mean": 0.9187357425689697, "rewards/meter/std": 0.09213165193796158, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7782436609268188, "rewards/repeat_soft/std": 0.08641798794269562, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.13715866208076477, "rewards/total_composite/mean": 0.6613816618919373, "rewards/total_composite/std": 0.27163782715797424, "reward": 0.6613816618919373, "reward_std": 0.27163782715797424, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06574160605669022, "sampling/sampling_logp_difference/max": 1.8433537483215332, "sampling/importance_sampling_ratio/min": 0.1582856923341751, "sampling/importance_sampling_ratio/mean": 1.008857011795044, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3576795198023319, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.05796498851850629, "clip_ratio/high_max": 0.05796498851850629, "clip_ratio/region_mean": 0.05796498851850629, "reward_total_mean": 0.6613816618919373, "reward_meter_mean": 0.9187357425689697, "reward_meter_std": 0.09213165193796158, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7782436609268188, "reward_repeat_soft_std": 0.08641798794269562, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.13715866208076477, "reward_total_composite_mean": 0.6613816618919373, "reward_total_composite_std": 0.27163782715797424} {"timestamp_utc": "2026-04-13T04:19:59Z", "mode": "train", "global_step": 2385, "epoch": 0.2395781014565545, "loss": -0.0209, "grad_norm": 11.858123779296875, "learning_rate": 2.7757575757575762e-06, "num_tokens": 4409928.0, "completions/mean_length": 56.75, "completions/min_length": 51.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.75, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9841405153274536, "rewards/meter/std": 0.00457889586687088, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9488848447799683, "rewards/repeat_soft/std": 0.0409424863755703, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8148767352104187, "rewards/total_composite/std": 0.00667211739346385, "reward": 0.8148767352104187, "reward_std": 0.006672120187431574, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10333018749952316, "sampling/sampling_logp_difference/max": 1.0254721641540527, "sampling/importance_sampling_ratio/min": 0.3586271107196808, "sampling/importance_sampling_ratio/mean": 1.0096688270568848, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.658342756330967, "clip_ratio/low_mean": 0.018625584663823247, "clip_ratio/low_min": 0.018625584663823247, "clip_ratio/high_mean": 0.09397117979824543, "clip_ratio/high_max": 0.09397117979824543, "clip_ratio/region_mean": 0.11259676446206868, "reward_total_mean": 0.8148767352104187, "reward_meter_mean": 0.9841405153274536, "reward_meter_std": 0.00457889586687088, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9488848447799683, "reward_repeat_soft_std": 0.0409424863755703, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8148767352104187, "reward_total_composite_std": 0.00667211739346385} {"timestamp_utc": "2026-04-13T04:20:06Z", "mode": "train", "global_step": 2386, "epoch": 0.23967855349070818, "loss": 0.0476, "grad_norm": 11.804705619812012, "learning_rate": 2.772727272727273e-06, "num_tokens": 4411577.0, "completions/mean_length": 55.125, "completions/min_length": 46.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.125, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9077647924423218, "rewards/meter/std": 0.11087192595005035, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9601547718048096, "rewards/repeat_soft/std": 0.03284064307808876, "rewards/judge_quality/mean": 0.6262500286102295, "rewards/judge_quality/std": 0.2144719660282135, "rewards/total_composite/mean": 0.8423846364021301, "rewards/total_composite/std": 0.07549537718296051, "reward": 0.8423846364021301, "reward_std": 0.07549536228179932, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13788747787475586, "sampling/sampling_logp_difference/max": 2.1530356407165527, "sampling/importance_sampling_ratio/min": 0.11613108962774277, "sampling/importance_sampling_ratio/mean": 1.0039299726486206, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.753820575773716, "clip_ratio/low_mean": 0.06876653339713812, "clip_ratio/low_min": 0.06876653339713812, "clip_ratio/high_mean": 0.06998860463500023, "clip_ratio/high_max": 0.06998860463500023, "clip_ratio/region_mean": 0.13875513803213835, "reward_total_mean": 0.8423846364021301, "reward_meter_mean": 0.9077647924423218, "reward_meter_std": 0.11087192595005035, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9601547718048096, "reward_repeat_soft_std": 0.03284064307808876, "reward_judge_quality_mean": 0.6262500286102295, "reward_judge_quality_std": 0.2144719660282135, "reward_total_composite_mean": 0.8423846364021301, "reward_total_composite_std": 0.07549537718296051} {"timestamp_utc": "2026-04-13T04:20:14Z", "mode": "train", "global_step": 2387, "epoch": 0.23977900552486187, "loss": 0.002, "grad_norm": 9.148175239562988, "learning_rate": 2.76969696969697e-06, "num_tokens": 4414071.0, "completions/mean_length": 137.75, "completions/min_length": 127.0, "completions/max_length": 156.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 137.75, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 156.0, "rewards/meter/mean": 0.9878226518630981, "rewards/meter/std": 0.011445592157542706, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.918464183807373, "rewards/repeat_soft/std": 0.08529206365346909, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.8022416234016418, "rewards/total_composite/std": 0.033372264355421066, "reward": 0.8022416234016418, "reward_std": 0.03337226063013077, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08004593849182129, "sampling/sampling_logp_difference/max": 1.6907110214233398, "sampling/importance_sampling_ratio/min": 0.26117220520973206, "sampling/importance_sampling_ratio/mean": 1.0135215520858765, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.478754386305809, "clip_ratio/low_mean": 0.01601245766505599, "clip_ratio/low_min": 0.01601245766505599, "clip_ratio/high_mean": 0.05405461369082332, "clip_ratio/high_max": 0.05405461369082332, "clip_ratio/region_mean": 0.0700670713558793, "reward_total_mean": 0.8022416234016418, "reward_meter_mean": 0.9878226518630981, "reward_meter_std": 0.011445592157542706, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.918464183807373, "reward_repeat_soft_std": 0.08529206365346909, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.8022416234016418, "reward_total_composite_std": 0.033372264355421066} {"timestamp_utc": "2026-04-13T04:20:25Z", "mode": "train", "global_step": 2388, "epoch": 0.23987945755901557, "loss": -0.1089, "grad_norm": 2.7642948627471924, "learning_rate": 2.766666666666667e-06, "num_tokens": 4415771.0, "completions/mean_length": 110.5, "completions/min_length": 46.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 53.142860412597656, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.6197270154953003, "rewards/meter/std": 0.43556714057922363, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.957767903804779, "rewards/repeat_soft/std": 0.06366148591041565, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.13274572789669037, "rewards/total_composite/mean": 0.6052476763725281, "rewards/total_composite/std": 0.29145532846450806, "reward": 0.6052476763725281, "reward_std": 0.29145532846450806, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.097467340528965, "sampling/sampling_logp_difference/max": 1.1875581741333008, "sampling/importance_sampling_ratio/min": 0.3049650192260742, "sampling/importance_sampling_ratio/mean": 0.9969063997268677, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4806158244609833, "clip_ratio/low_mean": 0.0264498433098197, "clip_ratio/low_min": 0.0264498433098197, "clip_ratio/high_mean": 0.07060703728348017, "clip_ratio/high_max": 0.07060703728348017, "clip_ratio/region_mean": 0.09705688059329987, "reward_total_mean": 0.6052476763725281, "reward_meter_mean": 0.6197270154953003, "reward_meter_std": 0.43556714057922363, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.957767903804779, "reward_repeat_soft_std": 0.06366148591041565, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.13274572789669037, "reward_total_composite_mean": 0.6052476763725281, "reward_total_composite_std": 0.29145532846450806} {"timestamp_utc": "2026-04-13T04:20:32Z", "mode": "train", "global_step": 2389, "epoch": 0.23997990959316925, "loss": -0.0118, "grad_norm": 8.246169090270996, "learning_rate": 2.763636363636364e-06, "num_tokens": 4417423.0, "completions/mean_length": 41.5, "completions/min_length": 37.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.5, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.945700466632843, "rewards/meter/std": 0.042210936546325684, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8592190146446228, "rewards/repeat_soft/std": 0.0968293622136116, "rewards/judge_quality/mean": 0.367499977350235, "rewards/judge_quality/std": 0.09808888286352158, "rewards/total_composite/mean": 0.7717370986938477, "rewards/total_composite/std": 0.0488579235970974, "reward": 0.7717370986938477, "reward_std": 0.0488579235970974, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07691163569688797, "sampling/sampling_logp_difference/max": 1.5227468013763428, "sampling/importance_sampling_ratio/min": 0.21811196208000183, "sampling/importance_sampling_ratio/mean": 1.0127902030944824, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3498742878437042, "clip_ratio/low_mean": 0.046607252676039934, "clip_ratio/low_min": 0.046607252676039934, "clip_ratio/high_mean": 0.04391662171110511, "clip_ratio/high_max": 0.04391662171110511, "clip_ratio/region_mean": 0.09052387438714504, "reward_total_mean": 0.7717370986938477, "reward_meter_mean": 0.945700466632843, "reward_meter_std": 0.042210936546325684, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8592190146446228, "reward_repeat_soft_std": 0.0968293622136116, "reward_judge_quality_mean": 0.367499977350235, "reward_judge_quality_std": 0.09808888286352158, "reward_total_composite_mean": 0.7717370986938477, "reward_total_composite_std": 0.0488579235970974} {"timestamp_utc": "2026-04-13T04:20:38Z", "mode": "train", "global_step": 2390, "epoch": 0.24008036162732296, "loss": 0.0347, "grad_norm": 15.132451057434082, "learning_rate": 2.760606060606061e-06, "num_tokens": 4419217.0, "completions/mean_length": 64.25, "completions/min_length": 59.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.25, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9616168737411499, "rewards/meter/std": 0.07789094746112823, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9611321091651917, "rewards/repeat_soft/std": 0.030025631189346313, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.0843462198972702, "rewards/total_composite/mean": 0.7943408489227295, "rewards/total_composite/std": 0.03687683120369911, "reward": 0.7943408489227295, "reward_std": 0.036876823753118515, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11359100043773651, "sampling/sampling_logp_difference/max": 1.6659317016601562, "sampling/importance_sampling_ratio/min": 0.1890144795179367, "sampling/importance_sampling_ratio/mean": 1.0050841569900513, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6517370492219925, "clip_ratio/low_mean": 0.030493980273604393, "clip_ratio/low_min": 0.030493980273604393, "clip_ratio/high_mean": 0.08849802799522877, "clip_ratio/high_max": 0.08849802799522877, "clip_ratio/region_mean": 0.11899200826883316, "reward_total_mean": 0.7943408489227295, "reward_meter_mean": 0.9616168737411499, "reward_meter_std": 0.07789094746112823, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9611321091651917, "reward_repeat_soft_std": 0.030025631189346313, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.0843462198972702, "reward_total_composite_mean": 0.7943408489227295, "reward_total_composite_std": 0.03687683120369911} {"timestamp_utc": "2026-04-13T04:20:50Z", "mode": "train", "global_step": 2391, "epoch": 0.24018081366147664, "loss": -0.1541, "grad_norm": 1.6875724792480469, "learning_rate": 2.7575757575757576e-06, "num_tokens": 4421060.0, "completions/mean_length": 183.375, "completions/min_length": 63.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 73.83333587646484, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.8429017066955566, "rewards/meter/std": 0.3425855338573456, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9593838453292847, "rewards/repeat_soft/std": 0.0539955273270607, "rewards/judge_quality/mean": 0.39000001549720764, "rewards/judge_quality/std": 0.27166154980659485, "rewards/total_composite/mean": 0.6263121366500854, "rewards/total_composite/std": 0.3908265233039856, "reward": 0.6263121366500854, "reward_std": 0.3908265233039856, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09724890440702438, "sampling/sampling_logp_difference/max": 1.5820388793945312, "sampling/importance_sampling_ratio/min": 0.20555557310581207, "sampling/importance_sampling_ratio/mean": 1.004427433013916, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.457167599350214, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08289411757141352, "clip_ratio/high_max": 0.08289411757141352, "clip_ratio/region_mean": 0.08289411757141352, "reward_total_mean": 0.6263121366500854, "reward_meter_mean": 0.8429017066955566, "reward_meter_std": 0.3425855338573456, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9593838453292847, "reward_repeat_soft_std": 0.0539955273270607, "reward_judge_quality_mean": 0.39000001549720764, "reward_judge_quality_std": 0.27166154980659485, "reward_total_composite_mean": 0.6263121366500854, "reward_total_composite_std": 0.3908265233039856} {"timestamp_utc": "2026-04-13T04:21:01Z", "mode": "train", "global_step": 2392, "epoch": 0.24028126569563032, "loss": -0.0911, "grad_norm": 2.5755395889282227, "learning_rate": 2.754545454545455e-06, "num_tokens": 4422434.0, "completions/mean_length": 92.75, "completions/min_length": 28.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 32.85714340209961, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.7500062584877014, "rewards/meter/std": 0.4369015097618103, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9508851170539856, "rewards/repeat_soft/std": 0.028487678617239, "rewards/judge_quality/mean": 0.39249998331069946, "rewards/judge_quality/std": 0.13905291259288788, "rewards/total_composite/mean": 0.6674551963806152, "rewards/total_composite/std": 0.30752140283584595, "reward": 0.6674551963806152, "reward_std": 0.30752137303352356, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11064609885215759, "sampling/sampling_logp_difference/max": 1.8927736282348633, "sampling/importance_sampling_ratio/min": 0.15065337717533112, "sampling/importance_sampling_ratio/mean": 1.0160290002822876, "sampling/importance_sampling_ratio/max": 1.800553798675537, "entropy": 0.7022255398333073, "clip_ratio/low_mean": 0.0036764706019312143, "clip_ratio/low_min": 0.0036764706019312143, "clip_ratio/high_mean": 0.061550953425467014, "clip_ratio/high_max": 0.061550953425467014, "clip_ratio/region_mean": 0.06522742402739823, "reward_total_mean": 0.6674551963806152, "reward_meter_mean": 0.7500062584877014, "reward_meter_std": 0.4369015097618103, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9508851170539856, "reward_repeat_soft_std": 0.028487678617239, "reward_judge_quality_mean": 0.39249998331069946, "reward_judge_quality_std": 0.13905291259288788, "reward_total_composite_mean": 0.6674551963806152, "reward_total_composite_std": 0.30752140283584595} {"timestamp_utc": "2026-04-13T04:21:09Z", "mode": "train", "global_step": 2393, "epoch": 0.24038171772978403, "loss": 0.045, "grad_norm": 6.256504535675049, "learning_rate": 2.7515151515151517e-06, "num_tokens": 4424818.0, "completions/mean_length": 122.0, "completions/min_length": 116.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.0, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9937293529510498, "rewards/meter/std": 0.00397126842290163, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8765937089920044, "rewards/repeat_soft/std": 0.04312301054596901, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.8670876026153564, "rewards/total_composite/std": 0.07990426570177078, "reward": 0.8670876026153564, "reward_std": 0.07990426570177078, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08035656064748764, "sampling/sampling_logp_difference/max": 1.7315080165863037, "sampling/importance_sampling_ratio/min": 0.17701725661754608, "sampling/importance_sampling_ratio/mean": 1.009107232093811, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4171230271458626, "clip_ratio/low_mean": 0.049006045795977116, "clip_ratio/low_min": 0.049006045795977116, "clip_ratio/high_mean": 0.027511932887136936, "clip_ratio/high_max": 0.027511932887136936, "clip_ratio/region_mean": 0.07651797868311405, "reward_total_mean": 0.8670876026153564, "reward_meter_mean": 0.9937293529510498, "reward_meter_std": 0.00397126842290163, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8765937089920044, "reward_repeat_soft_std": 0.04312301054596901, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.8670876026153564, "reward_total_composite_std": 0.07990426570177078} {"timestamp_utc": "2026-04-13T04:21:15Z", "mode": "train", "global_step": 2394, "epoch": 0.24048216976393771, "loss": 0.0457, "grad_norm": 14.354172706604004, "learning_rate": 2.7484848484848486e-06, "num_tokens": 4426249.0, "completions/mean_length": 25.875, "completions/min_length": 20.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.875, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.9746337532997131, "rewards/meter/std": 0.043535638600587845, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9585226774215698, "rewards/repeat_soft/std": 0.011249415576457977, "rewards/judge_quality/mean": 0.7987500429153442, "rewards/judge_quality/std": 0.22465451061725616, "rewards/total_composite/mean": 0.9240624308586121, "rewards/total_composite/std": 0.06676149368286133, "reward": 0.9240624308586121, "reward_std": 0.06676150113344193, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12824364006519318, "sampling/sampling_logp_difference/max": 1.7841529846191406, "sampling/importance_sampling_ratio/min": 0.16793926060199738, "sampling/importance_sampling_ratio/mean": 1.020650863647461, "sampling/importance_sampling_ratio/max": 1.7340887784957886, "entropy": 0.800406739115715, "clip_ratio/low_mean": 0.03916666749864817, "clip_ratio/low_min": 0.03916666749864817, "clip_ratio/high_mean": 0.07880731392651796, "clip_ratio/high_max": 0.07880731392651796, "clip_ratio/region_mean": 0.11797398142516613, "reward_total_mean": 0.9240624308586121, "reward_meter_mean": 0.9746337532997131, "reward_meter_std": 0.043535638600587845, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9585226774215698, "reward_repeat_soft_std": 0.011249415576457977, "reward_judge_quality_mean": 0.7987500429153442, "reward_judge_quality_std": 0.22465451061725616, "reward_total_composite_mean": 0.9240624308586121, "reward_total_composite_std": 0.06676149368286133} {"timestamp_utc": "2026-04-13T04:21:22Z", "mode": "train", "global_step": 2395, "epoch": 0.24058262179809142, "loss": -0.0099, "grad_norm": 4.451852798461914, "learning_rate": 2.7454545454545454e-06, "num_tokens": 4428497.0, "completions/mean_length": 130.0, "completions/min_length": 116.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 130.0, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.9601417779922485, "rewards/meter/std": 0.0745956152677536, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7463949918746948, "rewards/repeat_soft/std": 0.10800544172525406, "rewards/judge_quality/mean": 0.3474999964237213, "rewards/judge_quality/std": 0.11310549825429916, "rewards/total_composite/mean": 0.7609533071517944, "rewards/total_composite/std": 0.03904452919960022, "reward": 0.7609533071517944, "reward_std": 0.039044544100761414, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07397189736366272, "sampling/sampling_logp_difference/max": 2.8768153190612793, "sampling/importance_sampling_ratio/min": 0.05631381645798683, "sampling/importance_sampling_ratio/mean": 1.0059291124343872, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.454816572368145, "clip_ratio/low_mean": 0.03276082780212164, "clip_ratio/low_min": 0.03276082780212164, "clip_ratio/high_mean": 0.041423695627599955, "clip_ratio/high_max": 0.041423695627599955, "clip_ratio/region_mean": 0.0741845234297216, "reward_total_mean": 0.7609533071517944, "reward_meter_mean": 0.9601417779922485, "reward_meter_std": 0.0745956152677536, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7463949918746948, "reward_repeat_soft_std": 0.10800544172525406, "reward_judge_quality_mean": 0.3474999964237213, "reward_judge_quality_std": 0.11310549825429916, "reward_total_composite_mean": 0.7609533071517944, "reward_total_composite_std": 0.03904452919960022} {"timestamp_utc": "2026-04-13T04:21:29Z", "mode": "train", "global_step": 2396, "epoch": 0.2406830738322451, "loss": 0.0231, "grad_norm": 9.937352180480957, "learning_rate": 2.7424242424242426e-06, "num_tokens": 4430433.0, "completions/mean_length": 55.0, "completions/min_length": 49.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.0, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.984430193901062, "rewards/meter/std": 0.006986097898334265, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9804763793945312, "rewards/repeat_soft/std": 0.030340587720274925, "rewards/judge_quality/mean": 0.5674999952316284, "rewards/judge_quality/std": 0.21756774187088013, "rewards/total_composite/mean": 0.8612911701202393, "rewards/total_composite/std": 0.0663255825638771, "reward": 0.8612911701202393, "reward_std": 0.06632556766271591, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10304562002420425, "sampling/sampling_logp_difference/max": 1.7498571872711182, "sampling/importance_sampling_ratio/min": 0.17379875481128693, "sampling/importance_sampling_ratio/mean": 0.9994689226150513, "sampling/importance_sampling_ratio/max": 1.8324203491210938, "entropy": 0.5073417946696281, "clip_ratio/low_mean": 0.05859287828207016, "clip_ratio/low_min": 0.05859287828207016, "clip_ratio/high_mean": 0.023021885193884373, "clip_ratio/high_max": 0.023021885193884373, "clip_ratio/region_mean": 0.08161476347595453, "reward_total_mean": 0.8612911701202393, "reward_meter_mean": 0.984430193901062, "reward_meter_std": 0.006986097898334265, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9804763793945312, "reward_repeat_soft_std": 0.030340587720274925, "reward_judge_quality_mean": 0.5674999952316284, "reward_judge_quality_std": 0.21756774187088013, "reward_total_composite_mean": 0.8612911701202393, "reward_total_composite_std": 0.0663255825638771} {"timestamp_utc": "2026-04-13T04:21:37Z", "mode": "train", "global_step": 2397, "epoch": 0.24078352586639878, "loss": -0.0213, "grad_norm": 3.589263916015625, "learning_rate": 2.7393939393939395e-06, "num_tokens": 4433841.0, "completions/mean_length": 218.0, "completions/min_length": 204.0, "completions/max_length": 231.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 218.0, "completions/min_terminated_length": 204.0, "completions/max_terminated_length": 231.0, "rewards/meter/mean": 0.8661840558052063, "rewards/meter/std": 0.2492513209581375, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6547912359237671, "rewards/repeat_soft/std": 0.12372793257236481, "rewards/judge_quality/mean": 0.26749998331069946, "rewards/judge_quality/std": 0.10375107079744339, "rewards/total_composite/mean": 0.6792619824409485, "rewards/total_composite/std": 0.15021535754203796, "reward": 0.6792619824409485, "reward_std": 0.15021534264087677, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05838225781917572, "sampling/sampling_logp_difference/max": 1.4196009635925293, "sampling/importance_sampling_ratio/min": 0.24181050062179565, "sampling/importance_sampling_ratio/mean": 1.0087759494781494, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37310850992798805, "clip_ratio/low_mean": 0.010416666744276881, "clip_ratio/low_min": 0.010416666744276881, "clip_ratio/high_mean": 0.03986114216968417, "clip_ratio/high_max": 0.03986114216968417, "clip_ratio/region_mean": 0.05027780891396105, "reward_total_mean": 0.6792619824409485, "reward_meter_mean": 0.8661840558052063, "reward_meter_std": 0.2492513209581375, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6547912359237671, "reward_repeat_soft_std": 0.12372793257236481, "reward_judge_quality_mean": 0.26749998331069946, "reward_judge_quality_std": 0.10375107079744339, "reward_total_composite_mean": 0.6792619824409485, "reward_total_composite_std": 0.15021535754203796} {"timestamp_utc": "2026-04-13T04:21:44Z", "mode": "train", "global_step": 2398, "epoch": 0.2408839779005525, "loss": 0.0275, "grad_norm": 10.045125007629395, "learning_rate": 2.7363636363636363e-06, "num_tokens": 4435520.0, "completions/mean_length": 60.875, "completions/min_length": 54.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9460023641586304, "rewards/meter/std": 0.08299379795789719, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9693678617477417, "rewards/repeat_soft/std": 0.016121014952659607, "rewards/judge_quality/mean": 0.5575000047683716, "rewards/judge_quality/std": 0.19955308735370636, "rewards/total_composite/mean": 0.8398878574371338, "rewards/total_composite/std": 0.04666798934340477, "reward": 0.8398878574371338, "reward_std": 0.046667974442243576, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10153060406446457, "sampling/sampling_logp_difference/max": 1.76557195186615, "sampling/importance_sampling_ratio/min": 0.17108890414237976, "sampling/importance_sampling_ratio/mean": 0.9876875281333923, "sampling/importance_sampling_ratio/max": 1.692322850227356, "entropy": 0.48955435305833817, "clip_ratio/low_mean": 0.05765830026939511, "clip_ratio/low_min": 0.05765830026939511, "clip_ratio/high_mean": 0.028461960144340992, "clip_ratio/high_max": 0.028461960144340992, "clip_ratio/region_mean": 0.0861202604137361, "reward_total_mean": 0.8398878574371338, "reward_meter_mean": 0.9460023641586304, "reward_meter_std": 0.08299379795789719, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9693678617477417, "reward_repeat_soft_std": 0.016121014952659607, "reward_judge_quality_mean": 0.5575000047683716, "reward_judge_quality_std": 0.19955308735370636, "reward_total_composite_mean": 0.8398878574371338, "reward_total_composite_std": 0.04666798934340477} {"timestamp_utc": "2026-04-13T04:21:50Z", "mode": "train", "global_step": 2399, "epoch": 0.24098442993470617, "loss": 0.0194, "grad_norm": 12.824419021606445, "learning_rate": 2.7333333333333336e-06, "num_tokens": 4436896.0, "completions/mean_length": 30.0, "completions/min_length": 24.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.0, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9786720275878906, "rewards/meter/std": 0.03184375539422035, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.5674999952316284, "rewards/judge_quality/std": 0.21756774187088013, "rewards/total_composite/mean": 0.8569024205207825, "rewards/total_composite/std": 0.06798207759857178, "reward": 0.8569024205207825, "reward_std": 0.06798207014799118, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08937585353851318, "sampling/sampling_logp_difference/max": 1.2979779243469238, "sampling/importance_sampling_ratio/min": 0.2730834186077118, "sampling/importance_sampling_ratio/mean": 1.0004340410232544, "sampling/importance_sampling_ratio/max": 1.762143850326538, "entropy": 0.5210679247975349, "clip_ratio/low_mean": 0.04471049830317497, "clip_ratio/low_min": 0.04471049830317497, "clip_ratio/high_mean": 0.02557471301406622, "clip_ratio/high_max": 0.02557471301406622, "clip_ratio/region_mean": 0.07028521131724119, "reward_total_mean": 0.8569024205207825, "reward_meter_mean": 0.9786720275878906, "reward_meter_std": 0.03184375539422035, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.5674999952316284, "reward_judge_quality_std": 0.21756774187088013, "reward_total_composite_mean": 0.8569024205207825, "reward_total_composite_std": 0.06798207759857178} {"timestamp_utc": "2026-04-13T04:21:56Z", "mode": "train", "global_step": 2400, "epoch": 0.24108488196885988, "loss": 0.0249, "grad_norm": 7.380618572235107, "learning_rate": 2.7303030303030304e-06, "num_tokens": 4438887.0, "completions/mean_length": 70.875, "completions/min_length": 64.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.875, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9936805963516235, "rewards/meter/std": 0.0022035324946045876, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9485245943069458, "rewards/repeat_soft/std": 0.04168388620018959, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8180087208747864, "rewards/total_composite/std": 0.004034020006656647, "reward": 0.8180087208747864, "reward_std": 0.004034013953059912, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09097393602132797, "sampling/sampling_logp_difference/max": 1.5588641166687012, "sampling/importance_sampling_ratio/min": 0.2103748917579651, "sampling/importance_sampling_ratio/mean": 0.9978368282318115, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44087038934230804, "clip_ratio/low_mean": 0.02240704488940537, "clip_ratio/low_min": 0.02240704488940537, "clip_ratio/high_mean": 0.054586222395300865, "clip_ratio/high_max": 0.054586222395300865, "clip_ratio/region_mean": 0.07699326728470623, "reward_total_mean": 0.8180087208747864, "reward_meter_mean": 0.9936805963516235, "reward_meter_std": 0.0022035324946045876, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9485245943069458, "reward_repeat_soft_std": 0.04168388620018959, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8180087208747864, "reward_total_composite_std": 0.004034020006656647} {"timestamp_utc": "2026-04-13T04:23:00Z", "mode": "eval", "global_step": 2400, "epoch": 0.24108488196885988, "eval_loss": NaN, "eval_runtime": 63.3478, "eval_samples_per_second": 1.263, "eval_steps_per_second": 0.158, "eval_num_tokens": 4438887.0, "eval_completions/mean_length": 127.275, "eval_completions/min_length": 47.9, "eval_completions/max_length": 312.7, "eval_completions/clipped_ratio": 0.0625, "eval_completions/mean_terminated_length": 101.52440643310547, "eval_completions/min_terminated_length": 47.9, "eval_completions/max_terminated_length": 178.3, "eval_rewards/meter/mean": 0.8655049562454223, "eval_rewards/meter/std": 0.2312098227441311, "eval_rewards/count_adherence/mean": 0.9599999964237214, "eval_rewards/count_adherence/std": 0.08970049545168876, "eval_rewards/hard_gate/mean": 0.925, "eval_rewards/hard_gate/std": 0.18771235942840575, "eval_rewards/repeat_soft/mean": 0.9360461831092834, "eval_rewards/repeat_soft/std": 0.06359568871557712, "eval_rewards/judge_quality/mean": 0.3899999976158142, "eval_rewards/judge_quality/std": 0.15168366879224776, "eval_rewards/total_composite/mean": 0.7087890684604645, "eval_rewards/total_composite/std": 0.19664095602929593, "eval_reward": 0.7087890684604645, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.05758761763572693, "eval_sampling/sampling_logp_difference/max": 1.0191164016723633, "eval_sampling/importance_sampling_ratio/min": 0.3659633070230484, "eval_sampling/importance_sampling_ratio/mean": 1.0114330291748046, "eval_sampling/importance_sampling_ratio/max": 1.534163236618042, "eval_entropy": 0.6869515806436539, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7087890684604645, "eval_reward_meter_mean": 0.8655049562454223, "eval_reward_meter_std": 0.2312098227441311, "eval_reward_count_adherence_mean": 0.9599999964237214, "eval_reward_count_adherence_std": 0.08970049545168876, "eval_reward_hard_gate_mean": 0.925, "eval_reward_hard_gate_std": 0.18771235942840575, "eval_reward_repeat_soft_mean": 0.9360461831092834, "eval_reward_repeat_soft_std": 0.06359568871557712, "eval_reward_judge_quality_mean": 0.3899999976158142, "eval_reward_judge_quality_std": 0.15168366879224776, "eval_reward_total_composite_mean": 0.7087890684604645, "eval_reward_total_composite_std": 0.19664095602929593} {"timestamp_utc": "2026-04-13T04:23:09Z", "mode": "train", "global_step": 2401, "epoch": 0.24118533400301356, "loss": -0.0278, "grad_norm": 9.94188404083252, "learning_rate": 2.7272727272727272e-06, "num_tokens": 4440648.0, "completions/mean_length": 56.125, "completions/min_length": 50.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.125, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9848052263259888, "rewards/meter/std": 0.011910478584468365, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9755746126174927, "rewards/repeat_soft/std": 0.02668406069278717, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.8564698100090027, "rewards/total_composite/std": 0.06714136153459549, "reward": 0.8564698100090027, "reward_std": 0.06714136153459549, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09178552776575089, "sampling/sampling_logp_difference/max": 1.2722551822662354, "sampling/importance_sampling_ratio/min": 0.2801990211009979, "sampling/importance_sampling_ratio/mean": 1.0223511457443237, "sampling/importance_sampling_ratio/max": 1.9704294204711914, "entropy": 0.5440746992826462, "clip_ratio/low_mean": 0.07053062878549099, "clip_ratio/low_min": 0.07053062878549099, "clip_ratio/high_mean": 0.02271174918860197, "clip_ratio/high_max": 0.02271174918860197, "clip_ratio/region_mean": 0.09324237797409296, "reward_total_mean": 0.8564698100090027, "reward_meter_mean": 0.9848052263259888, "reward_meter_std": 0.011910478584468365, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9755746126174927, "reward_repeat_soft_std": 0.02668406069278717, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.8564698100090027, "reward_total_composite_std": 0.06714136153459549} {"timestamp_utc": "2026-04-13T04:23:16Z", "mode": "train", "global_step": 2402, "epoch": 0.24128578603716724, "loss": 0.0349, "grad_norm": 11.481598854064941, "learning_rate": 2.724242424242424e-06, "num_tokens": 4442327.0, "completions/mean_length": 56.875, "completions/min_length": 47.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.875, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.7995168566703796, "rewards/meter/std": 0.29366418719291687, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9637620449066162, "rewards/repeat_soft/std": 0.03256964683532715, "rewards/judge_quality/mean": 0.41749998927116394, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.7314087748527527, "rewards/total_composite/std": 0.1504307985305786, "reward": 0.7314087748527527, "reward_std": 0.15043078362941742, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1443522721529007, "sampling/sampling_logp_difference/max": 1.431748867034912, "sampling/importance_sampling_ratio/min": 0.23889076709747314, "sampling/importance_sampling_ratio/mean": 1.0000003576278687, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8765817508101463, "clip_ratio/low_mean": 0.02248008595779538, "clip_ratio/low_min": 0.02248008595779538, "clip_ratio/high_mean": 0.10508600063621998, "clip_ratio/high_max": 0.10508600063621998, "clip_ratio/region_mean": 0.12756608659401536, "reward_total_mean": 0.7314087748527527, "reward_meter_mean": 0.7995168566703796, "reward_meter_std": 0.29366418719291687, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9637620449066162, "reward_repeat_soft_std": 0.03256964683532715, "reward_judge_quality_mean": 0.41749998927116394, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.7314087748527527, "reward_total_composite_std": 0.1504307985305786} {"timestamp_utc": "2026-04-13T04:23:24Z", "mode": "train", "global_step": 2403, "epoch": 0.24138623807132095, "loss": 0.0239, "grad_norm": 8.14038372039795, "learning_rate": 2.7212121212121213e-06, "num_tokens": 4444878.0, "completions/mean_length": 131.875, "completions/min_length": 121.0, "completions/max_length": 147.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.875, "completions/min_terminated_length": 121.0, "completions/max_terminated_length": 147.0, "rewards/meter/mean": 0.8622517585754395, "rewards/meter/std": 0.22731724381446838, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8759043216705322, "rewards/repeat_soft/std": 0.030740909278392792, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7452287077903748, "rewards/total_composite/std": 0.09948767721652985, "reward": 0.7452287077903748, "reward_std": 0.09948766976594925, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09764259308576584, "sampling/sampling_logp_difference/max": 2.24869441986084, "sampling/importance_sampling_ratio/min": 0.10553692281246185, "sampling/importance_sampling_ratio/mean": 1.0139079093933105, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.42320092767477036, "clip_ratio/low_mean": 0.021938775666058064, "clip_ratio/low_min": 0.021938775666058064, "clip_ratio/high_mean": 0.06155281118117273, "clip_ratio/high_max": 0.06155281118117273, "clip_ratio/region_mean": 0.08349158684723079, "reward_total_mean": 0.7452287077903748, "reward_meter_mean": 0.8622517585754395, "reward_meter_std": 0.22731724381446838, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8759043216705322, "reward_repeat_soft_std": 0.030740909278392792, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7452287077903748, "reward_total_composite_std": 0.09948767721652985} {"timestamp_utc": "2026-04-13T04:23:36Z", "mode": "train", "global_step": 2404, "epoch": 0.24148669010547463, "loss": -0.2225, "grad_norm": 2.3117949962615967, "learning_rate": 2.718181818181818e-06, "num_tokens": 4447305.0, "completions/mean_length": 283.375, "completions/min_length": 140.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 146.1999969482422, "completions/min_terminated_length": 140.0, "completions/max_terminated_length": 151.0, "rewards/meter/mean": 0.6538940072059631, "rewards/meter/std": 0.34052035212516785, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.27124056220054626, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9414978623390198, "rewards/repeat_soft/std": 0.060131389647722244, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.2962111830711365, "rewards/total_composite/mean": 0.4714219570159912, "rewards/total_composite/std": 0.40129271149635315, "reward": 0.4714219570159912, "reward_std": 0.40129271149635315, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10159751027822495, "sampling/sampling_logp_difference/max": 1.5001935958862305, "sampling/importance_sampling_ratio/min": 0.22308696806430817, "sampling/importance_sampling_ratio/mean": 1.0170351266860962, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3969328999519348, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0676988223567605, "clip_ratio/high_max": 0.0676988223567605, "clip_ratio/region_mean": 0.0676988223567605, "reward_total_mean": 0.4714219570159912, "reward_meter_mean": 0.6538940072059631, "reward_meter_std": 0.34052035212516785, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.27124056220054626, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9414978623390198, "reward_repeat_soft_std": 0.060131389647722244, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.2962111830711365, "reward_total_composite_mean": 0.4714219570159912, "reward_total_composite_std": 0.40129271149635315} {"timestamp_utc": "2026-04-13T04:23:48Z", "mode": "train", "global_step": 2405, "epoch": 0.2415871421396283, "loss": -0.2263, "grad_norm": 1.6289619207382202, "learning_rate": 2.715151515151516e-06, "num_tokens": 4449762.0, "completions/mean_length": 189.125, "completions/min_length": 136.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 143.0, "completions/min_terminated_length": 136.0, "completions/max_terminated_length": 155.0, "rewards/meter/mean": 0.8449581861495972, "rewards/meter/std": 0.34124961495399475, "rewards/count_adherence/mean": 0.7250000238418579, "rewards/count_adherence/std": 0.2121320515871048, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.891873836517334, "rewards/repeat_soft/std": 0.08498871326446533, "rewards/judge_quality/mean": 0.3399999737739563, "rewards/judge_quality/std": 0.15052290260791779, "rewards/total_composite/mean": 0.6624805331230164, "rewards/total_composite/std": 0.2693658471107483, "reward": 0.6624805331230164, "reward_std": 0.2693658769130707, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08603654056787491, "sampling/sampling_logp_difference/max": 1.5473113059997559, "sampling/importance_sampling_ratio/min": 0.2128194123506546, "sampling/importance_sampling_ratio/mean": 1.0047043561935425, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45629480481147766, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0802091620862484, "clip_ratio/high_max": 0.0802091620862484, "clip_ratio/region_mean": 0.0802091620862484, "reward_total_mean": 0.6624805331230164, "reward_meter_mean": 0.8449581861495972, "reward_meter_std": 0.34124961495399475, "reward_count_adherence_mean": 0.7250000238418579, "reward_count_adherence_std": 0.2121320515871048, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.891873836517334, "reward_repeat_soft_std": 0.08498871326446533, "reward_judge_quality_mean": 0.3399999737739563, "reward_judge_quality_std": 0.15052290260791779, "reward_total_composite_mean": 0.6624805331230164, "reward_total_composite_std": 0.2693658471107483} {"timestamp_utc": "2026-04-13T04:23:56Z", "mode": "train", "global_step": 2406, "epoch": 0.24168759417378202, "loss": 0.0476, "grad_norm": 4.633444309234619, "learning_rate": 2.7121212121212127e-06, "num_tokens": 4452982.0, "completions/mean_length": 180.5, "completions/min_length": 173.0, "completions/max_length": 190.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 180.5, "completions/min_terminated_length": 173.0, "completions/max_terminated_length": 190.0, "rewards/meter/mean": 0.852014422416687, "rewards/meter/std": 0.2515667974948883, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6408504843711853, "rewards/repeat_soft/std": 0.07494300603866577, "rewards/judge_quality/mean": 0.30124998092651367, "rewards/judge_quality/std": 0.10398317128419876, "rewards/total_composite/mean": 0.6878665685653687, "rewards/total_composite/std": 0.13656528294086456, "reward": 0.6878665685653687, "reward_std": 0.13656529784202576, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07632258534431458, "sampling/sampling_logp_difference/max": 2.069047212600708, "sampling/importance_sampling_ratio/min": 0.12630607187747955, "sampling/importance_sampling_ratio/mean": 1.0042649507522583, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4418763294816017, "clip_ratio/low_mean": 0.015208566561341286, "clip_ratio/low_min": 0.015208566561341286, "clip_ratio/high_mean": 0.04729378176853061, "clip_ratio/high_max": 0.04729378176853061, "clip_ratio/region_mean": 0.06250234832987189, "reward_total_mean": 0.6878665685653687, "reward_meter_mean": 0.852014422416687, "reward_meter_std": 0.2515667974948883, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6408504843711853, "reward_repeat_soft_std": 0.07494300603866577, "reward_judge_quality_mean": 0.30124998092651367, "reward_judge_quality_std": 0.10398317128419876, "reward_total_composite_mean": 0.6878665685653687, "reward_total_composite_std": 0.13656528294086456} {"timestamp_utc": "2026-04-13T04:24:03Z", "mode": "train", "global_step": 2407, "epoch": 0.2417880462079357, "loss": 0.0231, "grad_norm": 9.388854026794434, "learning_rate": 2.7090909090909095e-06, "num_tokens": 4454711.0, "completions/mean_length": 64.125, "completions/min_length": 60.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.125, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9905609488487244, "rewards/meter/std": 0.004433033522218466, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9780408143997192, "rewards/repeat_soft/std": 0.017252309247851372, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.14603695273399353, "rewards/total_composite/mean": 0.8431814908981323, "rewards/total_composite/std": 0.04386002570390701, "reward": 0.8431814908981323, "reward_std": 0.043860021978616714, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10073131322860718, "sampling/sampling_logp_difference/max": 1.6749608516693115, "sampling/importance_sampling_ratio/min": 0.18731550872325897, "sampling/importance_sampling_ratio/mean": 1.0023587942123413, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5025118924677372, "clip_ratio/low_mean": 0.056448533199727535, "clip_ratio/low_min": 0.056448533199727535, "clip_ratio/high_mean": 0.021526054479181767, "clip_ratio/high_max": 0.021526054479181767, "clip_ratio/region_mean": 0.0779745876789093, "reward_total_mean": 0.8431814908981323, "reward_meter_mean": 0.9905609488487244, "reward_meter_std": 0.004433033522218466, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9780408143997192, "reward_repeat_soft_std": 0.017252309247851372, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.14603695273399353, "reward_total_composite_mean": 0.8431814908981323, "reward_total_composite_std": 0.04386002570390701} {"timestamp_utc": "2026-04-13T04:24:09Z", "mode": "train", "global_step": 2408, "epoch": 0.2418884982420894, "loss": -0.0224, "grad_norm": 7.99906063079834, "learning_rate": 2.7060606060606063e-06, "num_tokens": 4456343.0, "completions/mean_length": 55.0, "completions/min_length": 49.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.0, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9805049300193787, "rewards/meter/std": 0.01529120933264494, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9089842438697815, "rewards/repeat_soft/std": 0.08627494424581528, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.8051255941390991, "rewards/total_composite/std": 0.02781411074101925, "reward": 0.8051255941390991, "reward_std": 0.027814103290438652, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10281851887702942, "sampling/sampling_logp_difference/max": 1.382659912109375, "sampling/importance_sampling_ratio/min": 0.2509102523326874, "sampling/importance_sampling_ratio/mean": 1.0056499242782593, "sampling/importance_sampling_ratio/max": 1.8476476669311523, "entropy": 0.6442204564809799, "clip_ratio/low_mean": 0.03913473803550005, "clip_ratio/low_min": 0.03913473803550005, "clip_ratio/high_mean": 0.06203383393585682, "clip_ratio/high_max": 0.06203383393585682, "clip_ratio/region_mean": 0.10116857197135687, "reward_total_mean": 0.8051255941390991, "reward_meter_mean": 0.9805049300193787, "reward_meter_std": 0.01529120933264494, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9089842438697815, "reward_repeat_soft_std": 0.08627494424581528, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.8051255941390991, "reward_total_composite_std": 0.02781411074101925} {"timestamp_utc": "2026-04-13T04:24:20Z", "mode": "train", "global_step": 2409, "epoch": 0.2419889502762431, "loss": -0.0898, "grad_norm": 1.6340807676315308, "learning_rate": 2.7030303030303036e-06, "num_tokens": 4457856.0, "completions/mean_length": 155.125, "completions/min_length": 32.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 36.16666793823242, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.8457698225975037, "rewards/meter/std": 0.2738134264945984, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9582511782646179, "rewards/repeat_soft/std": 0.007705728989094496, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.1738995909690857, "rewards/total_composite/mean": 0.5704690217971802, "rewards/total_composite/std": 0.369295209646225, "reward": 0.5704690217971802, "reward_std": 0.369295209646225, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11328897625207901, "sampling/sampling_logp_difference/max": 2.2132511138916016, "sampling/importance_sampling_ratio/min": 0.10934457927942276, "sampling/importance_sampling_ratio/mean": 1.0212262868881226, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6406972035765648, "clip_ratio/low_mean": 0.013888888992369175, "clip_ratio/low_min": 0.013888888992369175, "clip_ratio/high_mean": 0.05202107154764235, "clip_ratio/high_max": 0.05202107154764235, "clip_ratio/region_mean": 0.06590996054001153, "reward_total_mean": 0.5704690217971802, "reward_meter_mean": 0.8457698225975037, "reward_meter_std": 0.2738134264945984, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9582511782646179, "reward_repeat_soft_std": 0.007705728989094496, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.1738995909690857, "reward_total_composite_mean": 0.5704690217971802, "reward_total_composite_std": 0.369295209646225} {"timestamp_utc": "2026-04-13T04:24:31Z", "mode": "train", "global_step": 2410, "epoch": 0.24208940231039677, "loss": -0.1022, "grad_norm": 1.900059461593628, "learning_rate": 2.7000000000000004e-06, "num_tokens": 4459339.0, "completions/mean_length": 95.375, "completions/min_length": 32.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 35.85714340209961, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.8642439842224121, "rewards/meter/std": 0.2796519100666046, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9522446393966675, "rewards/repeat_soft/std": 0.019564032554626465, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.6643802523612976, "rewards/total_composite/std": 0.29496335983276367, "reward": 0.6643802523612976, "reward_std": 0.2949633300304413, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08972134441137314, "sampling/sampling_logp_difference/max": 0.8655911684036255, "sampling/importance_sampling_ratio/min": 0.4208027422428131, "sampling/importance_sampling_ratio/mean": 1.002767562866211, "sampling/importance_sampling_ratio/max": 1.4389978647232056, "entropy": 0.6011215001344681, "clip_ratio/low_mean": 0.0173611119389534, "clip_ratio/low_min": 0.0173611119389534, "clip_ratio/high_mean": 0.10137104522436857, "clip_ratio/high_max": 0.10137104522436857, "clip_ratio/region_mean": 0.11873215716332197, "reward_total_mean": 0.6643802523612976, "reward_meter_mean": 0.8642439842224121, "reward_meter_std": 0.2796519100666046, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9522446393966675, "reward_repeat_soft_std": 0.019564032554626465, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.6643802523612976, "reward_total_composite_std": 0.29496335983276367} {"timestamp_utc": "2026-04-13T04:24:37Z", "mode": "train", "global_step": 2411, "epoch": 0.24218985434455048, "loss": 0.0032, "grad_norm": 16.57874870300293, "learning_rate": 2.6969696969696972e-06, "num_tokens": 4460774.0, "completions/mean_length": 28.375, "completions/min_length": 26.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.375, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.9823909997940063, "rewards/meter/std": 0.02432149089872837, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7443709373474121, "rewards/repeat_soft/std": 0.09581231325864792, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7958880662918091, "rewards/total_composite/std": 0.007326413877308369, "reward": 0.7958880662918091, "reward_std": 0.007326409220695496, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09203922748565674, "sampling/sampling_logp_difference/max": 1.016922950744629, "sampling/importance_sampling_ratio/min": 0.4579525291919708, "sampling/importance_sampling_ratio/mean": 0.9975036978721619, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4209502302110195, "clip_ratio/low_mean": 0.01841168152168393, "clip_ratio/low_min": 0.01841168152168393, "clip_ratio/high_mean": 0.06530172564089298, "clip_ratio/high_max": 0.06530172564089298, "clip_ratio/region_mean": 0.08371340716257691, "reward_total_mean": 0.7958880662918091, "reward_meter_mean": 0.9823909997940063, "reward_meter_std": 0.02432149089872837, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7443709373474121, "reward_repeat_soft_std": 0.09581231325864792, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7958880662918091, "reward_total_composite_std": 0.007326413877308369} {"timestamp_utc": "2026-04-13T04:24:49Z", "mode": "train", "global_step": 2412, "epoch": 0.24229030637870416, "loss": -0.1147, "grad_norm": 3.5236024856567383, "learning_rate": 2.6939393939393945e-06, "num_tokens": 4462350.0, "completions/mean_length": 106.0, "completions/min_length": 37.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 48.000003814697266, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.7545256614685059, "rewards/meter/std": 0.30628690123558044, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9851295351982117, "rewards/repeat_soft/std": 0.020171666517853737, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.2436625212430954, "rewards/total_composite/mean": 0.6256667375564575, "rewards/total_composite/std": 0.3018818795681, "reward": 0.6256667375564575, "reward_std": 0.3018818795681, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11017748713493347, "sampling/sampling_logp_difference/max": 2.190227508544922, "sampling/importance_sampling_ratio/min": 0.11189128458499908, "sampling/importance_sampling_ratio/mean": 1.0053892135620117, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5680685415863991, "clip_ratio/low_mean": 0.014694149140268564, "clip_ratio/low_min": 0.014694149140268564, "clip_ratio/high_mean": 0.06815696647390723, "clip_ratio/high_max": 0.06815696647390723, "clip_ratio/region_mean": 0.0828511156141758, "reward_total_mean": 0.6256667375564575, "reward_meter_mean": 0.7545256614685059, "reward_meter_std": 0.30628690123558044, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9851295351982117, "reward_repeat_soft_std": 0.020171666517853737, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.2436625212430954, "reward_total_composite_mean": 0.6256667375564575, "reward_total_composite_std": 0.3018818795681} {"timestamp_utc": "2026-04-13T04:24:55Z", "mode": "train", "global_step": 2413, "epoch": 0.24239075841285787, "loss": -0.0157, "grad_norm": 12.573569297790527, "learning_rate": 2.6909090909090913e-06, "num_tokens": 4464079.0, "completions/mean_length": 46.125, "completions/min_length": 42.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.125, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.4347187280654907, "rewards/meter/std": 0.36283549666404724, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9334736466407776, "rewards/repeat_soft/std": 0.08516387641429901, "rewards/judge_quality/mean": 0.5012500286102295, "rewards/judge_quality/std": 0.16974246501922607, "rewards/total_composite/mean": 0.5893458127975464, "rewards/total_composite/std": 0.15247902274131775, "reward": 0.5893458127975464, "reward_std": 0.15247900784015656, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10340685397386551, "sampling/sampling_logp_difference/max": 1.8603668212890625, "sampling/importance_sampling_ratio/min": 0.15561552345752716, "sampling/importance_sampling_ratio/mean": 0.9938706755638123, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5256172232329845, "clip_ratio/low_mean": 0.058339925948530436, "clip_ratio/low_min": 0.058339925948530436, "clip_ratio/high_mean": 0.04376465268433094, "clip_ratio/high_max": 0.04376465268433094, "clip_ratio/region_mean": 0.10210457863286138, "reward_total_mean": 0.5893458127975464, "reward_meter_mean": 0.4347187280654907, "reward_meter_std": 0.36283549666404724, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9334736466407776, "reward_repeat_soft_std": 0.08516387641429901, "reward_judge_quality_mean": 0.5012500286102295, "reward_judge_quality_std": 0.16974246501922607, "reward_total_composite_mean": 0.5893458127975464, "reward_total_composite_std": 0.15247902274131775} {"timestamp_utc": "2026-04-13T04:25:02Z", "mode": "train", "global_step": 2414, "epoch": 0.24249121044701155, "loss": 0.02, "grad_norm": 10.145086288452148, "learning_rate": 2.687878787878788e-06, "num_tokens": 4465815.0, "completions/mean_length": 54.0, "completions/min_length": 49.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9891195297241211, "rewards/meter/std": 0.0020356562454253435, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9766961932182312, "rewards/repeat_soft/std": 0.023279571905732155, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.8386484384536743, "rewards/total_composite/std": 0.0528310164809227, "reward": 0.8386484384536743, "reward_std": 0.0528310164809227, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09957768023014069, "sampling/sampling_logp_difference/max": 1.4092512130737305, "sampling/importance_sampling_ratio/min": 0.2443261593580246, "sampling/importance_sampling_ratio/mean": 1.002703070640564, "sampling/importance_sampling_ratio/max": 1.9123718738555908, "entropy": 0.6360092610120773, "clip_ratio/low_mean": 0.084181800018996, "clip_ratio/low_min": 0.084181800018996, "clip_ratio/high_mean": 0.004807692486792803, "clip_ratio/high_max": 0.004807692486792803, "clip_ratio/region_mean": 0.0889894925057888, "reward_total_mean": 0.8386484384536743, "reward_meter_mean": 0.9891195297241211, "reward_meter_std": 0.0020356562454253435, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9766961932182312, "reward_repeat_soft_std": 0.023279571905732155, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.8386484384536743, "reward_total_composite_std": 0.0528310164809227} {"timestamp_utc": "2026-04-13T04:25:09Z", "mode": "train", "global_step": 2415, "epoch": 0.24259166248116523, "loss": 0.132, "grad_norm": 9.766316413879395, "learning_rate": 2.684848484848485e-06, "num_tokens": 4467979.0, "completions/mean_length": 88.5, "completions/min_length": 76.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 88.5, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.9199616312980652, "rewards/meter/std": 0.15307284891605377, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9745802283287048, "rewards/repeat_soft/std": 0.01957709528505802, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.8124407529830933, "rewards/total_composite/std": 0.12693190574645996, "reward": 0.8124407529830933, "reward_std": 0.12693189084529877, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1056390106678009, "sampling/sampling_logp_difference/max": 4.8829145431518555, "sampling/importance_sampling_ratio/min": 0.00757490424439311, "sampling/importance_sampling_ratio/mean": 1.0134847164154053, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5216451808810234, "clip_ratio/low_mean": 0.04259259346872568, "clip_ratio/low_min": 0.04259259346872568, "clip_ratio/high_mean": 0.06037774449214339, "clip_ratio/high_max": 0.06037774449214339, "clip_ratio/region_mean": 0.10297033796086907, "reward_total_mean": 0.8124407529830933, "reward_meter_mean": 0.9199616312980652, "reward_meter_std": 0.15307284891605377, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9745802283287048, "reward_repeat_soft_std": 0.01957709528505802, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.8124407529830933, "reward_total_composite_std": 0.12693190574645996} {"timestamp_utc": "2026-04-13T04:25:16Z", "mode": "train", "global_step": 2416, "epoch": 0.24269211451531894, "loss": 0.0493, "grad_norm": 9.989828109741211, "learning_rate": 2.6818181818181822e-06, "num_tokens": 4470626.0, "completions/mean_length": 135.875, "completions/min_length": 120.0, "completions/max_length": 149.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 135.875, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 149.0, "rewards/meter/mean": 0.8274889588356018, "rewards/meter/std": 0.2649464011192322, "rewards/count_adherence/mean": 0.8541666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8554421663284302, "rewards/repeat_soft/std": 0.08927693217992783, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.7307892441749573, "rewards/total_composite/std": 0.135855033993721, "reward": 0.7307892441749573, "reward_std": 0.135855033993721, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12582309544086456, "sampling/sampling_logp_difference/max": 2.2462100982666016, "sampling/importance_sampling_ratio/min": 0.10579943656921387, "sampling/importance_sampling_ratio/mean": 1.008908987045288, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47029899060726166, "clip_ratio/low_mean": 0.03176032844930887, "clip_ratio/low_min": 0.03176032844930887, "clip_ratio/high_mean": 0.06152855604887009, "clip_ratio/high_max": 0.06152855604887009, "clip_ratio/region_mean": 0.09328888449817896, "reward_total_mean": 0.7307892441749573, "reward_meter_mean": 0.8274889588356018, "reward_meter_std": 0.2649464011192322, "reward_count_adherence_mean": 0.8541666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8554421663284302, "reward_repeat_soft_std": 0.08927693217992783, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.7307892441749573, "reward_total_composite_std": 0.135855033993721} {"timestamp_utc": "2026-04-13T04:25:23Z", "mode": "train", "global_step": 2417, "epoch": 0.24279256654947262, "loss": 0.0318, "grad_norm": 7.2433671951293945, "learning_rate": 2.678787878787879e-06, "num_tokens": 4473001.0, "completions/mean_length": 94.875, "completions/min_length": 88.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.875, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.814944326877594, "rewards/meter/std": 0.12894068658351898, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.967837929725647, "rewards/repeat_soft/std": 0.018277611583471298, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.7695087194442749, "rewards/total_composite/std": 0.05491297319531441, "reward": 0.7695087194442749, "reward_std": 0.054912954568862915, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08696573972702026, "sampling/sampling_logp_difference/max": 1.3905892372131348, "sampling/importance_sampling_ratio/min": 0.24892859160900116, "sampling/importance_sampling_ratio/mean": 0.9982596039772034, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4439450204372406, "clip_ratio/low_mean": 0.04093425627797842, "clip_ratio/low_min": 0.04093425627797842, "clip_ratio/high_mean": 0.052188914734870195, "clip_ratio/high_max": 0.052188914734870195, "clip_ratio/region_mean": 0.09312317101284862, "reward_total_mean": 0.7695087194442749, "reward_meter_mean": 0.814944326877594, "reward_meter_std": 0.12894068658351898, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.967837929725647, "reward_repeat_soft_std": 0.018277611583471298, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.7695087194442749, "reward_total_composite_std": 0.05491297319531441} {"timestamp_utc": "2026-04-13T04:25:29Z", "mode": "train", "global_step": 2418, "epoch": 0.24289301858362633, "loss": 0.0655, "grad_norm": 10.91202449798584, "learning_rate": 2.675757575757576e-06, "num_tokens": 4474780.0, "completions/mean_length": 54.375, "completions/min_length": 49.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.375, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9824473857879639, "rewards/meter/std": 0.01605728454887867, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9577220678329468, "rewards/repeat_soft/std": 0.0386054590344429, "rewards/judge_quality/mean": 0.5774999856948853, "rewards/judge_quality/std": 0.2988908290863037, "rewards/total_composite/mean": 0.8611235618591309, "rewards/total_composite/std": 0.09003522247076035, "reward": 0.8611235618591309, "reward_std": 0.09003520756959915, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10101477056741714, "sampling/sampling_logp_difference/max": 3.3447518348693848, "sampling/importance_sampling_ratio/min": 0.03526896610856056, "sampling/importance_sampling_ratio/mean": 0.9986464381217957, "sampling/importance_sampling_ratio/max": 1.8311330080032349, "entropy": 0.559972982853651, "clip_ratio/low_mean": 0.05547602707520127, "clip_ratio/low_min": 0.05547602707520127, "clip_ratio/high_mean": 0.04696355480700731, "clip_ratio/high_max": 0.04696355480700731, "clip_ratio/region_mean": 0.10243958188220859, "reward_total_mean": 0.8611235618591309, "reward_meter_mean": 0.9824473857879639, "reward_meter_std": 0.01605728454887867, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9577220678329468, "reward_repeat_soft_std": 0.0386054590344429, "reward_judge_quality_mean": 0.5774999856948853, "reward_judge_quality_std": 0.2988908290863037, "reward_total_composite_mean": 0.8611235618591309, "reward_total_composite_std": 0.09003522247076035} {"timestamp_utc": "2026-04-13T04:25:37Z", "mode": "train", "global_step": 2419, "epoch": 0.24299347061778, "loss": 0.0334, "grad_norm": 4.938743591308594, "learning_rate": 2.6727272727272727e-06, "num_tokens": 4477414.0, "completions/mean_length": 146.25, "completions/min_length": 142.0, "completions/max_length": 153.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 146.25, "completions/min_terminated_length": 142.0, "completions/max_terminated_length": 153.0, "rewards/meter/mean": 0.8366414904594421, "rewards/meter/std": 0.2270188331604004, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.605190634727478, "rewards/repeat_soft/std": 0.09985100477933884, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.6875077486038208, "rewards/total_composite/std": 0.11065588891506195, "reward": 0.6875077486038208, "reward_std": 0.11065588146448135, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.060277506709098816, "sampling/sampling_logp_difference/max": 2.110931873321533, "sampling/importance_sampling_ratio/min": 0.12112503498792648, "sampling/importance_sampling_ratio/mean": 1.0078171491622925, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3194757867604494, "clip_ratio/low_mean": 0.020143149886280298, "clip_ratio/low_min": 0.020143149886280298, "clip_ratio/high_mean": 0.025187105406075716, "clip_ratio/high_max": 0.025187105406075716, "clip_ratio/region_mean": 0.045330255292356014, "reward_total_mean": 0.6875077486038208, "reward_meter_mean": 0.8366414904594421, "reward_meter_std": 0.2270188331604004, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.605190634727478, "reward_repeat_soft_std": 0.09985100477933884, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.6875077486038208, "reward_total_composite_std": 0.11065588891506195} {"timestamp_utc": "2026-04-13T04:25:43Z", "mode": "train", "global_step": 2420, "epoch": 0.2430939226519337, "loss": 0.003, "grad_norm": 12.896089553833008, "learning_rate": 2.66969696969697e-06, "num_tokens": 4478936.0, "completions/mean_length": 46.25, "completions/min_length": 42.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.25, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.7825660705566406, "rewards/meter/std": 0.2959570586681366, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9946645498275757, "rewards/repeat_soft/std": 0.00758193526417017, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.786121129989624, "rewards/total_composite/std": 0.18251845240592957, "reward": 0.786121129989624, "reward_std": 0.18251845240592957, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09484273940324783, "sampling/sampling_logp_difference/max": 0.9792861938476562, "sampling/importance_sampling_ratio/min": 0.38063815236091614, "sampling/importance_sampling_ratio/mean": 0.9940335750579834, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.425707571208477, "clip_ratio/low_mean": 0.04758963920176029, "clip_ratio/low_min": 0.04758963920176029, "clip_ratio/high_mean": 0.0709158917888999, "clip_ratio/high_max": 0.0709158917888999, "clip_ratio/region_mean": 0.11850553099066019, "reward_total_mean": 0.786121129989624, "reward_meter_mean": 0.7825660705566406, "reward_meter_std": 0.2959570586681366, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9946645498275757, "reward_repeat_soft_std": 0.00758193526417017, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.786121129989624, "reward_total_composite_std": 0.18251845240592957} {"timestamp_utc": "2026-04-13T04:25:50Z", "mode": "train", "global_step": 2421, "epoch": 0.2431943746860874, "loss": 0.0038, "grad_norm": 12.925252914428711, "learning_rate": 2.666666666666667e-06, "num_tokens": 4480724.0, "completions/mean_length": 54.5, "completions/min_length": 51.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.5, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.858267068862915, "rewards/meter/std": 0.2558090388774872, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9640058875083923, "rewards/repeat_soft/std": 0.036192990839481354, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.7556207776069641, "rewards/total_composite/std": 0.11804857105016708, "reward": 0.7556207776069641, "reward_std": 0.11804857105016708, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09817396849393845, "sampling/sampling_logp_difference/max": 1.2568410634994507, "sampling/importance_sampling_ratio/min": 0.36362993717193604, "sampling/importance_sampling_ratio/mean": 1.006485104560852, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5706259943544865, "clip_ratio/low_mean": 0.023796791210770607, "clip_ratio/low_min": 0.023796791210770607, "clip_ratio/high_mean": 0.09313822910189629, "clip_ratio/high_max": 0.09313822910189629, "clip_ratio/region_mean": 0.11693502031266689, "reward_total_mean": 0.7556207776069641, "reward_meter_mean": 0.858267068862915, "reward_meter_std": 0.2558090388774872, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9640058875083923, "reward_repeat_soft_std": 0.036192990839481354, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.7556207776069641, "reward_total_composite_std": 0.11804857105016708} {"timestamp_utc": "2026-04-13T04:25:56Z", "mode": "train", "global_step": 2422, "epoch": 0.24329482672024108, "loss": 0.0234, "grad_norm": 8.862018585205078, "learning_rate": 2.6636363636363637e-06, "num_tokens": 4482476.0, "completions/mean_length": 59.0, "completions/min_length": 56.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.0, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9889893531799316, "rewards/meter/std": 0.002275317208841443, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9605381488800049, "rewards/repeat_soft/std": 0.026670601218938828, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8193490505218506, "rewards/total_composite/std": 0.005599416326731443, "reward": 0.8193490505218506, "reward_std": 0.005599421449005604, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0814911499619484, "sampling/sampling_logp_difference/max": 1.1759865283966064, "sampling/importance_sampling_ratio/min": 0.30851447582244873, "sampling/importance_sampling_ratio/mean": 1.011455774307251, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4184563457965851, "clip_ratio/low_mean": 0.03605100768618286, "clip_ratio/low_min": 0.03605100768618286, "clip_ratio/high_mean": 0.042382351122796535, "clip_ratio/high_max": 0.042382351122796535, "clip_ratio/region_mean": 0.07843335880897939, "reward_total_mean": 0.8193490505218506, "reward_meter_mean": 0.9889893531799316, "reward_meter_std": 0.002275317208841443, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9605381488800049, "reward_repeat_soft_std": 0.026670601218938828, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8193490505218506, "reward_total_composite_std": 0.005599416326731443} {"timestamp_utc": "2026-04-13T04:26:03Z", "mode": "train", "global_step": 2423, "epoch": 0.2433952787543948, "loss": 0.0483, "grad_norm": 8.320180892944336, "learning_rate": 2.660606060606061e-06, "num_tokens": 4484559.0, "completions/mean_length": 95.375, "completions/min_length": 87.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.375, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.9291207790374756, "rewards/meter/std": 0.1482038050889969, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9705899357795715, "rewards/repeat_soft/std": 0.016474220901727676, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7847883701324463, "rewards/total_composite/std": 0.06648281961679459, "reward": 0.7847883701324463, "reward_std": 0.06648281216621399, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10992518067359924, "sampling/sampling_logp_difference/max": 1.4040708541870117, "sampling/importance_sampling_ratio/min": 0.24559514224529266, "sampling/importance_sampling_ratio/mean": 1.0107216835021973, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5535605624318123, "clip_ratio/low_mean": 0.01759904995560646, "clip_ratio/low_min": 0.01759904995560646, "clip_ratio/high_mean": 0.09403384756296873, "clip_ratio/high_max": 0.09403384756296873, "clip_ratio/region_mean": 0.11163289751857519, "reward_total_mean": 0.7847883701324463, "reward_meter_mean": 0.9291207790374756, "reward_meter_std": 0.1482038050889969, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9705899357795715, "reward_repeat_soft_std": 0.016474220901727676, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7847883701324463, "reward_total_composite_std": 0.06648281961679459} {"timestamp_utc": "2026-04-13T04:26:10Z", "mode": "train", "global_step": 2424, "epoch": 0.24349573078854847, "loss": 0.0575, "grad_norm": 46.58006286621094, "learning_rate": 2.6575757575757577e-06, "num_tokens": 4487233.0, "completions/mean_length": 147.25, "completions/min_length": 137.0, "completions/max_length": 157.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 147.25, "completions/min_terminated_length": 137.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.9936869144439697, "rewards/meter/std": 0.004558468237519264, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9030612707138062, "rewards/repeat_soft/std": 0.0548376627266407, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.7969651818275452, "rewards/total_composite/std": 0.030933456495404243, "reward": 0.7969651818275452, "reward_std": 0.030933476984500885, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09367102384567261, "sampling/sampling_logp_difference/max": 2.880642890930176, "sampling/importance_sampling_ratio/min": 0.056098684668540955, "sampling/importance_sampling_ratio/mean": 1.0005643367767334, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5446949116885662, "clip_ratio/low_mean": 0.015774283558130264, "clip_ratio/low_min": 0.015774283558130264, "clip_ratio/high_mean": 0.06967266229912639, "clip_ratio/high_max": 0.06967266229912639, "clip_ratio/region_mean": 0.08544694585725665, "reward_total_mean": 0.7969651818275452, "reward_meter_mean": 0.9936869144439697, "reward_meter_std": 0.004558468237519264, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9030612707138062, "reward_repeat_soft_std": 0.0548376627266407, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.7969651818275452, "reward_total_composite_std": 0.030933456495404243} {"timestamp_utc": "2026-04-13T04:26:17Z", "mode": "train", "global_step": 2425, "epoch": 0.24359618282270215, "loss": 0.0173, "grad_norm": 11.271186828613281, "learning_rate": 2.6545454545454546e-06, "num_tokens": 4488943.0, "completions/mean_length": 61.75, "completions/min_length": 52.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.75, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.885547399520874, "rewards/meter/std": 0.2710934281349182, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9864815473556519, "rewards/repeat_soft/std": 0.01187772024422884, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.25150617957115173, "rewards/total_composite/mean": 0.8042694926261902, "rewards/total_composite/std": 0.15037985146045685, "reward": 0.8042694926261902, "reward_std": 0.15037985146045685, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1104232519865036, "sampling/sampling_logp_difference/max": 1.947214126586914, "sampling/importance_sampling_ratio/min": 0.14267098903656006, "sampling/importance_sampling_ratio/mean": 1.0220271348953247, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.59475377202034, "clip_ratio/low_mean": 0.028391062282025814, "clip_ratio/low_min": 0.028391062282025814, "clip_ratio/high_mean": 0.06828743172809482, "clip_ratio/high_max": 0.06828743172809482, "clip_ratio/region_mean": 0.09667849401012063, "reward_total_mean": 0.8042694926261902, "reward_meter_mean": 0.885547399520874, "reward_meter_std": 0.2710934281349182, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9864815473556519, "reward_repeat_soft_std": 0.01187772024422884, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.25150617957115173, "reward_total_composite_mean": 0.8042694926261902, "reward_total_composite_std": 0.15037985146045685} {"timestamp_utc": "2026-04-13T04:26:23Z", "mode": "train", "global_step": 2426, "epoch": 0.24369663485685586, "loss": 0.0178, "grad_norm": 10.878361701965332, "learning_rate": 2.6515151515151514e-06, "num_tokens": 4490632.0, "completions/mean_length": 55.125, "completions/min_length": 47.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.125, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.5117013454437256, "rewards/meter/std": 0.3428478538990021, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9593480825424194, "rewards/repeat_soft/std": 0.04209980368614197, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.6794504523277283, "rewards/total_composite/std": 0.20810763537883759, "reward": 0.6794504523277283, "reward_std": 0.20810763537883759, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1240212470293045, "sampling/sampling_logp_difference/max": 2.3098344802856445, "sampling/importance_sampling_ratio/min": 0.09927768260240555, "sampling/importance_sampling_ratio/mean": 1.0232113599777222, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6005854345858097, "clip_ratio/low_mean": 0.07684459816664457, "clip_ratio/low_min": 0.07684459816664457, "clip_ratio/high_mean": 0.0501840952783823, "clip_ratio/high_max": 0.0501840952783823, "clip_ratio/region_mean": 0.12702869344502687, "reward_total_mean": 0.6794504523277283, "reward_meter_mean": 0.5117013454437256, "reward_meter_std": 0.3428478538990021, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9593480825424194, "reward_repeat_soft_std": 0.04209980368614197, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.6794504523277283, "reward_total_composite_std": 0.20810763537883759} {"timestamp_utc": "2026-04-13T04:26:29Z", "mode": "train", "global_step": 2427, "epoch": 0.24379708689100954, "loss": 0.0029, "grad_norm": 9.596781730651855, "learning_rate": 2.6484848484848487e-06, "num_tokens": 4492384.0, "completions/mean_length": 56.0, "completions/min_length": 52.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.0, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9716958999633789, "rewards/meter/std": 0.041919659823179245, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9849621057510376, "rewards/repeat_soft/std": 0.016853639855980873, "rewards/judge_quality/mean": 0.5674999952316284, "rewards/judge_quality/std": 0.21756774187088013, "rewards/total_composite/mean": 0.8560093641281128, "rewards/total_composite/std": 0.07042060047388077, "reward": 0.8560093641281128, "reward_std": 0.07042059302330017, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1016862541437149, "sampling/sampling_logp_difference/max": 2.2591867446899414, "sampling/importance_sampling_ratio/min": 0.10443538427352905, "sampling/importance_sampling_ratio/mean": 1.0067400932312012, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6182467341423035, "clip_ratio/low_mean": 0.06440590228885412, "clip_ratio/low_min": 0.06440590228885412, "clip_ratio/high_mean": 0.026819923892617226, "clip_ratio/high_max": 0.026819923892617226, "clip_ratio/region_mean": 0.09122582618147135, "reward_total_mean": 0.8560093641281128, "reward_meter_mean": 0.9716958999633789, "reward_meter_std": 0.041919659823179245, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9849621057510376, "reward_repeat_soft_std": 0.016853639855980873, "reward_judge_quality_mean": 0.5674999952316284, "reward_judge_quality_std": 0.21756774187088013, "reward_total_composite_mean": 0.8560093641281128, "reward_total_composite_std": 0.07042060047388077} {"timestamp_utc": "2026-04-13T04:26:38Z", "mode": "train", "global_step": 2428, "epoch": 0.24389753892516322, "loss": 0.024, "grad_norm": 4.223513603210449, "learning_rate": 2.6454545454545455e-06, "num_tokens": 4495486.0, "completions/mean_length": 175.75, "completions/min_length": 156.0, "completions/max_length": 211.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 175.75, "completions/min_terminated_length": 156.0, "completions/max_terminated_length": 211.0, "rewards/meter/mean": 0.9657323360443115, "rewards/meter/std": 0.03771665692329407, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8406927585601807, "rewards/repeat_soft/std": 0.131138414144516, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7743988037109375, "rewards/total_composite/std": 0.021235469728708267, "reward": 0.7743988037109375, "reward_std": 0.02123546414077282, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0842076912522316, "sampling/sampling_logp_difference/max": 2.799168586730957, "sampling/importance_sampling_ratio/min": 0.06086064130067825, "sampling/importance_sampling_ratio/mean": 1.0002317428588867, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45445236936211586, "clip_ratio/low_mean": 0.023073386400938034, "clip_ratio/low_min": 0.023073386400938034, "clip_ratio/high_mean": 0.04968973156064749, "clip_ratio/high_max": 0.04968973156064749, "clip_ratio/region_mean": 0.07276311796158552, "reward_total_mean": 0.7743988037109375, "reward_meter_mean": 0.9657323360443115, "reward_meter_std": 0.03771665692329407, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8406927585601807, "reward_repeat_soft_std": 0.131138414144516, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7743988037109375, "reward_total_composite_std": 0.021235469728708267} {"timestamp_utc": "2026-04-13T04:26:49Z", "mode": "train", "global_step": 2429, "epoch": 0.24399799095931693, "loss": -0.1763, "grad_norm": 1.6387660503387451, "learning_rate": 2.6424242424242423e-06, "num_tokens": 4497363.0, "completions/mean_length": 139.625, "completions/min_length": 77.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 86.42857360839844, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.8577080368995667, "rewards/meter/std": 0.3468305766582489, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9602335691452026, "rewards/repeat_soft/std": 0.047734178602695465, "rewards/judge_quality/mean": 0.5225000381469727, "rewards/judge_quality/std": 0.2682083249092102, "rewards/total_composite/mean": 0.7556169629096985, "rewards/total_composite/std": 0.3109769821166992, "reward": 0.7556169629096985, "reward_std": 0.3109769821166992, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11610250920057297, "sampling/sampling_logp_difference/max": 1.8900542259216309, "sampling/importance_sampling_ratio/min": 0.15106360614299774, "sampling/importance_sampling_ratio/mean": 1.0029278993606567, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5014433935284615, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09442471014335752, "clip_ratio/high_max": 0.09442471014335752, "clip_ratio/region_mean": 0.09442471014335752, "reward_total_mean": 0.7556169629096985, "reward_meter_mean": 0.8577080368995667, "reward_meter_std": 0.3468305766582489, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9602335691452026, "reward_repeat_soft_std": 0.047734178602695465, "reward_judge_quality_mean": 0.5225000381469727, "reward_judge_quality_std": 0.2682083249092102, "reward_total_composite_mean": 0.7556169629096985, "reward_total_composite_std": 0.3109769821166992} {"timestamp_utc": "2026-04-13T04:26:57Z", "mode": "train", "global_step": 2430, "epoch": 0.2440984429934706, "loss": 0.0324, "grad_norm": 4.843735694885254, "learning_rate": 2.6393939393939396e-06, "num_tokens": 4499975.0, "completions/mean_length": 148.5, "completions/min_length": 138.0, "completions/max_length": 163.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 148.5, "completions/min_terminated_length": 138.0, "completions/max_terminated_length": 163.0, "rewards/meter/mean": 0.9672786593437195, "rewards/meter/std": 0.050026554614305496, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8263390064239502, "rewards/repeat_soft/std": 0.10891783982515335, "rewards/judge_quality/mean": 0.2549999952316284, "rewards/judge_quality/std": 0.111867256462574, "rewards/total_composite/mean": 0.7444092631340027, "rewards/total_composite/std": 0.04277435317635536, "reward": 0.7444092631340027, "reward_std": 0.042774349451065063, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08575811982154846, "sampling/sampling_logp_difference/max": 1.6829824447631836, "sampling/importance_sampling_ratio/min": 0.18581895530223846, "sampling/importance_sampling_ratio/mean": 1.0019631385803223, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45661837235093117, "clip_ratio/low_mean": 0.020519201643764973, "clip_ratio/low_min": 0.020519201643764973, "clip_ratio/high_mean": 0.05335971154272556, "clip_ratio/high_max": 0.05335971154272556, "clip_ratio/region_mean": 0.07387891318649054, "reward_total_mean": 0.7444092631340027, "reward_meter_mean": 0.9672786593437195, "reward_meter_std": 0.050026554614305496, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8263390064239502, "reward_repeat_soft_std": 0.10891783982515335, "reward_judge_quality_mean": 0.2549999952316284, "reward_judge_quality_std": 0.111867256462574, "reward_total_composite_mean": 0.7444092631340027, "reward_total_composite_std": 0.04277435317635536} {"timestamp_utc": "2026-04-13T04:27:05Z", "mode": "train", "global_step": 2431, "epoch": 0.24419889502762432, "loss": 0.0135, "grad_norm": 4.845733642578125, "learning_rate": 2.6363636363636364e-06, "num_tokens": 4503019.0, "completions/mean_length": 172.5, "completions/min_length": 151.0, "completions/max_length": 180.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 172.5, "completions/min_terminated_length": 151.0, "completions/max_terminated_length": 180.0, "rewards/meter/mean": 0.9940513372421265, "rewards/meter/std": 0.0018447955371811986, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9140241146087646, "rewards/repeat_soft/std": 0.05670106038451195, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.785475492477417, "rewards/total_composite/std": 0.028671622276306152, "reward": 0.785475492477417, "reward_std": 0.028671611100435257, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08608763664960861, "sampling/sampling_logp_difference/max": 2.094764232635498, "sampling/importance_sampling_ratio/min": 0.12309926003217697, "sampling/importance_sampling_ratio/mean": 1.0106137990951538, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5269725807011127, "clip_ratio/low_mean": 0.04597767745144665, "clip_ratio/low_min": 0.04597767745144665, "clip_ratio/high_mean": 0.021364837884902954, "clip_ratio/high_max": 0.021364837884902954, "clip_ratio/region_mean": 0.0673425153363496, "reward_total_mean": 0.785475492477417, "reward_meter_mean": 0.9940513372421265, "reward_meter_std": 0.0018447955371811986, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9140241146087646, "reward_repeat_soft_std": 0.05670106038451195, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.785475492477417, "reward_total_composite_std": 0.028671622276306152} {"timestamp_utc": "2026-04-13T04:27:11Z", "mode": "train", "global_step": 2432, "epoch": 0.244299347061778, "loss": 0.0108, "grad_norm": 9.412483215332031, "learning_rate": 2.6333333333333332e-06, "num_tokens": 4504834.0, "completions/mean_length": 63.875, "completions/min_length": 55.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.875, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9173546433448792, "rewards/meter/std": 0.12297496944665909, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9037300944328308, "rewards/repeat_soft/std": 0.051293518394231796, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7600575685501099, "rewards/total_composite/std": 0.053944751620292664, "reward": 0.7600575685501099, "reward_std": 0.05394475534558296, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09435056895017624, "sampling/sampling_logp_difference/max": 1.7027342319488525, "sampling/importance_sampling_ratio/min": 0.18218471109867096, "sampling/importance_sampling_ratio/mean": 1.0101605653762817, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5358737036585808, "clip_ratio/low_mean": 0.05022396193817258, "clip_ratio/low_min": 0.05022396193817258, "clip_ratio/high_mean": 0.05618802923709154, "clip_ratio/high_max": 0.05618802923709154, "clip_ratio/region_mean": 0.10641199117526412, "reward_total_mean": 0.7600575685501099, "reward_meter_mean": 0.9173546433448792, "reward_meter_std": 0.12297496944665909, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9037300944328308, "reward_repeat_soft_std": 0.051293518394231796, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7600575685501099, "reward_total_composite_std": 0.053944751620292664} {"timestamp_utc": "2026-04-13T04:27:17Z", "mode": "train", "global_step": 2433, "epoch": 0.24439979909593168, "loss": 0.0287, "grad_norm": 14.777425765991211, "learning_rate": 2.63030303030303e-06, "num_tokens": 4506292.0, "completions/mean_length": 30.25, "completions/min_length": 28.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.25, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9544612169265747, "rewards/meter/std": 0.05638968572020531, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9233396053314209, "rewards/repeat_soft/std": 0.05170498788356781, "rewards/judge_quality/mean": 0.44624999165534973, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8057165145874023, "rewards/total_composite/std": 0.025962743908166885, "reward": 0.8057165145874023, "reward_std": 0.025962738320231438, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09872196614742279, "sampling/sampling_logp_difference/max": 1.379591464996338, "sampling/importance_sampling_ratio/min": 0.2516813576221466, "sampling/importance_sampling_ratio/mean": 0.9825376272201538, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44282612204551697, "clip_ratio/low_mean": 0.008198924828320742, "clip_ratio/low_min": 0.008198924828320742, "clip_ratio/high_mean": 0.058212332893162966, "clip_ratio/high_max": 0.058212332893162966, "clip_ratio/region_mean": 0.06641125772148371, "reward_total_mean": 0.8057165145874023, "reward_meter_mean": 0.9544612169265747, "reward_meter_std": 0.05638968572020531, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9233396053314209, "reward_repeat_soft_std": 0.05170498788356781, "reward_judge_quality_mean": 0.44624999165534973, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8057165145874023, "reward_total_composite_std": 0.025962743908166885} {"timestamp_utc": "2026-04-13T04:27:29Z", "mode": "train", "global_step": 2434, "epoch": 0.2445002511300854, "loss": -0.0911, "grad_norm": 2.47636079788208, "learning_rate": 2.6272727272727278e-06, "num_tokens": 4507729.0, "completions/mean_length": 88.625, "completions/min_length": 24.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 28.142858505249023, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.8837410807609558, "rewards/meter/std": 0.15404944121837616, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9622748494148254, "rewards/repeat_soft/std": 0.0006367546156980097, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.2876474857330322, "rewards/total_composite/mean": 0.729648232460022, "rewards/total_composite/std": 0.3092782497406006, "reward": 0.729648232460022, "reward_std": 0.3092782199382782, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1341559886932373, "sampling/sampling_logp_difference/max": 1.2123827934265137, "sampling/importance_sampling_ratio/min": 0.2974875867366791, "sampling/importance_sampling_ratio/mean": 0.9938321709632874, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6341387778520584, "clip_ratio/low_mean": 0.02083333395421505, "clip_ratio/low_min": 0.02083333395421505, "clip_ratio/high_mean": 0.09806674905121326, "clip_ratio/high_max": 0.09806674905121326, "clip_ratio/region_mean": 0.11890008300542831, "reward_total_mean": 0.729648232460022, "reward_meter_mean": 0.8837410807609558, "reward_meter_std": 0.15404944121837616, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9622748494148254, "reward_repeat_soft_std": 0.0006367546156980097, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.2876474857330322, "reward_total_composite_mean": 0.729648232460022, "reward_total_composite_std": 0.3092782497406006} {"timestamp_utc": "2026-04-13T04:27:40Z", "mode": "train", "global_step": 2435, "epoch": 0.24460070316423907, "loss": -0.1119, "grad_norm": 1.4919155836105347, "learning_rate": 2.6242424242424246e-06, "num_tokens": 4509355.0, "completions/mean_length": 164.25, "completions/min_length": 44.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 48.333335876464844, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.7624291777610779, "rewards/meter/std": 0.36808350682258606, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9669879674911499, "rewards/repeat_soft/std": 0.025193704292178154, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.22025959193706512, "rewards/total_composite/mean": 0.6574994921684265, "rewards/total_composite/std": 0.3153645098209381, "reward": 0.6574994921684265, "reward_std": 0.3153644800186157, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1309879571199417, "sampling/sampling_logp_difference/max": 2.602464199066162, "sampling/importance_sampling_ratio/min": 0.07409077882766724, "sampling/importance_sampling_ratio/mean": 0.9959161281585693, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4490785710513592, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09337978530675173, "clip_ratio/high_max": 0.09337978530675173, "clip_ratio/region_mean": 0.09337978530675173, "reward_total_mean": 0.6574994921684265, "reward_meter_mean": 0.7624291777610779, "reward_meter_std": 0.36808350682258606, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9669879674911499, "reward_repeat_soft_std": 0.025193704292178154, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.22025959193706512, "reward_total_composite_mean": 0.6574994921684265, "reward_total_composite_std": 0.3153645098209381} {"timestamp_utc": "2026-04-13T04:27:48Z", "mode": "train", "global_step": 2436, "epoch": 0.24470115519839278, "loss": 0.0039, "grad_norm": 4.449008941650391, "learning_rate": 2.621212121212122e-06, "num_tokens": 4511581.0, "completions/mean_length": 115.25, "completions/min_length": 96.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.25, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9802858829498291, "rewards/meter/std": 0.013181586749851704, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.851606011390686, "rewards/repeat_soft/std": 0.09876227378845215, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.8146642446517944, "rewards/total_composite/std": 0.05881073698401451, "reward": 0.8146642446517944, "reward_std": 0.058810748159885406, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08655564486980438, "sampling/sampling_logp_difference/max": 1.2372970581054688, "sampling/importance_sampling_ratio/min": 0.2901674509048462, "sampling/importance_sampling_ratio/mean": 1.009765625, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4940572455525398, "clip_ratio/low_mean": 0.07179427053779364, "clip_ratio/low_min": 0.07179427053779364, "clip_ratio/high_mean": 0.009297520853579044, "clip_ratio/high_max": 0.009297520853579044, "clip_ratio/region_mean": 0.08109179139137268, "reward_total_mean": 0.8146642446517944, "reward_meter_mean": 0.9802858829498291, "reward_meter_std": 0.013181586749851704, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.851606011390686, "reward_repeat_soft_std": 0.09876227378845215, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.8146642446517944, "reward_total_composite_std": 0.05881073698401451} {"timestamp_utc": "2026-04-13T04:27:56Z", "mode": "train", "global_step": 2437, "epoch": 0.24480160723254646, "loss": 0.0487, "grad_norm": 7.872771263122559, "learning_rate": 2.6181818181818187e-06, "num_tokens": 4513934.0, "completions/mean_length": 118.125, "completions/min_length": 97.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.125, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.9669578075408936, "rewards/meter/std": 0.03829289227724075, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9710665941238403, "rewards/repeat_soft/std": 0.022382119670510292, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.12631450593471527, "rewards/total_composite/mean": 0.7816126346588135, "rewards/total_composite/std": 0.04067464545369148, "reward": 0.7816126346588135, "reward_std": 0.040674641728401184, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11464936286211014, "sampling/sampling_logp_difference/max": 2.1869468688964844, "sampling/importance_sampling_ratio/min": 0.11225896328687668, "sampling/importance_sampling_ratio/mean": 1.0088152885437012, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6332475021481514, "clip_ratio/low_mean": 0.05510716140270233, "clip_ratio/low_min": 0.05510716140270233, "clip_ratio/high_mean": 0.05023879650980234, "clip_ratio/high_max": 0.05023879650980234, "clip_ratio/region_mean": 0.10534595791250467, "reward_total_mean": 0.7816126346588135, "reward_meter_mean": 0.9669578075408936, "reward_meter_std": 0.03829289227724075, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9710665941238403, "reward_repeat_soft_std": 0.022382119670510292, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.12631450593471527, "reward_total_composite_mean": 0.7816126346588135, "reward_total_composite_std": 0.04067464545369148} {"timestamp_utc": "2026-04-13T04:28:02Z", "mode": "train", "global_step": 2438, "epoch": 0.24490205926670014, "loss": 0.0346, "grad_norm": 11.283024787902832, "learning_rate": 2.6151515151515155e-06, "num_tokens": 4515514.0, "completions/mean_length": 35.5, "completions/min_length": 33.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.5, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.9871872663497925, "rewards/meter/std": 0.011252658441662788, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8187342882156372, "rewards/total_composite/std": 0.006055417004972696, "reward": 0.8187342882156372, "reward_std": 0.00605541979894042, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08530712127685547, "sampling/sampling_logp_difference/max": 1.2856218814849854, "sampling/importance_sampling_ratio/min": 0.2764785885810852, "sampling/importance_sampling_ratio/mean": 1.0173263549804688, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6104023680090904, "clip_ratio/low_mean": 0.020467836409807205, "clip_ratio/low_min": 0.020467836409807205, "clip_ratio/high_mean": 0.05295690521597862, "clip_ratio/high_max": 0.05295690521597862, "clip_ratio/region_mean": 0.07342474162578583, "reward_total_mean": 0.8187342882156372, "reward_meter_mean": 0.9871872663497925, "reward_meter_std": 0.011252658441662788, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8187342882156372, "reward_total_composite_std": 0.006055417004972696} {"timestamp_utc": "2026-04-13T04:28:08Z", "mode": "train", "global_step": 2439, "epoch": 0.24500251130085385, "loss": 0.0207, "grad_norm": 6.970745086669922, "learning_rate": 2.6121212121212123e-06, "num_tokens": 4517282.0, "completions/mean_length": 70.0, "completions/min_length": 66.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9874272346496582, "rewards/meter/std": 0.004684466868638992, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.997655987739563, "rewards/repeat_soft/std": 0.002478762762621045, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.838857889175415, "rewards/total_composite/std": 0.053745511919260025, "reward": 0.838857889175415, "reward_std": 0.053745515644550323, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08612359315156937, "sampling/sampling_logp_difference/max": 1.4556519985198975, "sampling/importance_sampling_ratio/min": 0.2332482486963272, "sampling/importance_sampling_ratio/mean": 1.0135124921798706, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5111536979675293, "clip_ratio/low_mean": 0.08921676781028509, "clip_ratio/low_min": 0.08921676781028509, "clip_ratio/high_mean": 0.009057970717549324, "clip_ratio/high_max": 0.009057970717549324, "clip_ratio/region_mean": 0.09827473852783442, "reward_total_mean": 0.838857889175415, "reward_meter_mean": 0.9874272346496582, "reward_meter_std": 0.004684466868638992, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.997655987739563, "reward_repeat_soft_std": 0.002478762762621045, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.838857889175415, "reward_total_composite_std": 0.053745511919260025} {"timestamp_utc": "2026-04-13T04:28:15Z", "mode": "train", "global_step": 2440, "epoch": 0.24510296333500753, "loss": 0.0022, "grad_norm": 6.984889507293701, "learning_rate": 2.6090909090909096e-06, "num_tokens": 4518986.0, "completions/mean_length": 46.0, "completions/min_length": 39.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.0, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.9226459264755249, "rewards/meter/std": 0.02624588832259178, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9887923002243042, "rewards/repeat_soft/std": 0.013033751398324966, "rewards/judge_quality/mean": 0.5049999952316284, "rewards/judge_quality/std": 0.16801361739635468, "rewards/total_composite/mean": 0.8155698776245117, "rewards/total_composite/std": 0.057969409972429276, "reward": 0.8155698776245117, "reward_std": 0.057969409972429276, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07605685293674469, "sampling/sampling_logp_difference/max": 0.8530545234680176, "sampling/importance_sampling_ratio/min": 0.4261113703250885, "sampling/importance_sampling_ratio/mean": 1.00997793674469, "sampling/importance_sampling_ratio/max": 1.7394684553146362, "entropy": 0.46696562692523, "clip_ratio/low_mean": 0.04687609733082354, "clip_ratio/low_min": 0.04687609733082354, "clip_ratio/high_mean": 0.005319148767739534, "clip_ratio/high_max": 0.005319148767739534, "clip_ratio/region_mean": 0.052195246098563075, "reward_total_mean": 0.8155698776245117, "reward_meter_mean": 0.9226459264755249, "reward_meter_std": 0.02624588832259178, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9887923002243042, "reward_repeat_soft_std": 0.013033751398324966, "reward_judge_quality_mean": 0.5049999952316284, "reward_judge_quality_std": 0.16801361739635468, "reward_total_composite_mean": 0.8155698776245117, "reward_total_composite_std": 0.057969409972429276} {"timestamp_utc": "2026-04-13T04:28:21Z", "mode": "train", "global_step": 2441, "epoch": 0.24520341536916124, "loss": 0.0026, "grad_norm": 9.312705039978027, "learning_rate": 2.6060606060606064e-06, "num_tokens": 4520626.0, "completions/mean_length": 45.0, "completions/min_length": 43.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.0, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9433877468109131, "rewards/meter/std": 0.01851245015859604, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.798633873462677, "rewards/repeat_soft/std": 0.125315323472023, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.7785128355026245, "rewards/total_composite/std": 0.014321922324597836, "reward": 0.7785128355026245, "reward_std": 0.014321924187242985, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08436508476734161, "sampling/sampling_logp_difference/max": 1.3080625534057617, "sampling/importance_sampling_ratio/min": 0.2703433334827423, "sampling/importance_sampling_ratio/mean": 1.0021555423736572, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4372076913714409, "clip_ratio/low_mean": 0.03841350646689534, "clip_ratio/low_min": 0.03841350646689534, "clip_ratio/high_mean": 0.05836676899343729, "clip_ratio/high_max": 0.05836676899343729, "clip_ratio/region_mean": 0.09678027546033263, "reward_total_mean": 0.7785128355026245, "reward_meter_mean": 0.9433877468109131, "reward_meter_std": 0.01851245015859604, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.798633873462677, "reward_repeat_soft_std": 0.125315323472023, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.7785128355026245, "reward_total_composite_std": 0.014321922324597836} {"timestamp_utc": "2026-04-13T04:28:28Z", "mode": "train", "global_step": 2442, "epoch": 0.24530386740331492, "loss": 0.034, "grad_norm": 8.711477279663086, "learning_rate": 2.6030303030303033e-06, "num_tokens": 4522367.0, "completions/mean_length": 60.625, "completions/min_length": 53.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.625, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9892083406448364, "rewards/meter/std": 0.01008673943579197, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9840995073318481, "rewards/repeat_soft/std": 0.016193071380257607, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8218036890029907, "rewards/total_composite/std": 0.004238337278366089, "reward": 0.8218036890029907, "reward_std": 0.004238345194607973, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11428912729024887, "sampling/sampling_logp_difference/max": 1.4097919464111328, "sampling/importance_sampling_ratio/min": 0.24419407546520233, "sampling/importance_sampling_ratio/mean": 1.0058705806732178, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6464804485440254, "clip_ratio/low_mean": 0.07008619513362646, "clip_ratio/low_min": 0.07008619513362646, "clip_ratio/high_mean": 0.034953704103827477, "clip_ratio/high_max": 0.034953704103827477, "clip_ratio/region_mean": 0.10503989923745394, "reward_total_mean": 0.8218036890029907, "reward_meter_mean": 0.9892083406448364, "reward_meter_std": 0.01008673943579197, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9840995073318481, "reward_repeat_soft_std": 0.016193071380257607, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8218036890029907, "reward_total_composite_std": 0.004238337278366089} {"timestamp_utc": "2026-04-13T04:28:34Z", "mode": "train", "global_step": 2443, "epoch": 0.2454043194374686, "loss": 0.0479, "grad_norm": 13.915403366088867, "learning_rate": 2.6e-06, "num_tokens": 4524167.0, "completions/mean_length": 67.0, "completions/min_length": 61.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9853103160858154, "rewards/meter/std": 0.005246529821306467, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9609934687614441, "rewards/repeat_soft/std": 0.03962237760424614, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720350325107574, "rewards/total_composite/mean": 0.7534551620483398, "rewards/total_composite/std": 0.31168586015701294, "reward": 0.7534551620483398, "reward_std": 0.31168583035469055, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1199207529425621, "sampling/sampling_logp_difference/max": 3.1183674335479736, "sampling/importance_sampling_ratio/min": 0.04422931745648384, "sampling/importance_sampling_ratio/mean": 1.0074095726013184, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7169651761651039, "clip_ratio/low_mean": 0.01883561722934246, "clip_ratio/low_min": 0.01883561722934246, "clip_ratio/high_mean": 0.10739176347851753, "clip_ratio/high_max": 0.10739176347851753, "clip_ratio/region_mean": 0.12622738070786, "reward_total_mean": 0.7534551620483398, "reward_meter_mean": 0.9853103160858154, "reward_meter_std": 0.005246529821306467, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9609934687614441, "reward_repeat_soft_std": 0.03962237760424614, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720350325107574, "reward_total_composite_mean": 0.7534551620483398, "reward_total_composite_std": 0.31168586015701294} {"timestamp_utc": "2026-04-13T04:28:42Z", "mode": "train", "global_step": 2444, "epoch": 0.2455047714716223, "loss": -0.0018, "grad_norm": 6.326920509338379, "learning_rate": 2.5969696969696973e-06, "num_tokens": 4526455.0, "completions/mean_length": 118.0, "completions/min_length": 107.0, "completions/max_length": 127.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.0, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.49506229162216187, "rewards/meter/std": 0.2194945216178894, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9922667741775513, "rewards/repeat_soft/std": 0.004188511520624161, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.5172274708747864, "rewards/total_composite/std": 0.23033641278743744, "reward": 0.5172274708747864, "reward_std": 0.23033641278743744, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10259300470352173, "sampling/sampling_logp_difference/max": 2.3003058433532715, "sampling/importance_sampling_ratio/min": 0.1002281904220581, "sampling/importance_sampling_ratio/mean": 0.9998924732208252, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5193218104541302, "clip_ratio/low_mean": 0.034522161819040775, "clip_ratio/low_min": 0.034522161819040775, "clip_ratio/high_mean": 0.07295139133930206, "clip_ratio/high_max": 0.07295139133930206, "clip_ratio/region_mean": 0.10747355315834284, "reward_total_mean": 0.5172274708747864, "reward_meter_mean": 0.49506229162216187, "reward_meter_std": 0.2194945216178894, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9922667741775513, "reward_repeat_soft_std": 0.004188511520624161, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.5172274708747864, "reward_total_composite_std": 0.23033641278743744} {"timestamp_utc": "2026-04-13T04:28:48Z", "mode": "train", "global_step": 2445, "epoch": 0.245605223505776, "loss": 0.0424, "grad_norm": 9.5670747756958, "learning_rate": 2.593939393939394e-06, "num_tokens": 4528191.0, "completions/mean_length": 52.0, "completions/min_length": 49.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.0, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9921952486038208, "rewards/meter/std": 0.0020601593423634768, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9673808813095093, "rewards/repeat_soft/std": 0.02818349003791809, "rewards/judge_quality/mean": 0.6187499761581421, "rewards/judge_quality/std": 0.24976776540279388, "rewards/total_composite/mean": 0.8788509368896484, "rewards/total_composite/std": 0.07591696828603745, "reward": 0.8788509368896484, "reward_std": 0.07591696828603745, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10153915733098984, "sampling/sampling_logp_difference/max": 1.7320170402526855, "sampling/importance_sampling_ratio/min": 0.17692716419696808, "sampling/importance_sampling_ratio/mean": 1.0064561367034912, "sampling/importance_sampling_ratio/max": 1.843458652496338, "entropy": 0.5887462832033634, "clip_ratio/low_mean": 0.04776273528113961, "clip_ratio/low_min": 0.04776273528113961, "clip_ratio/high_mean": 0.031737884506583214, "clip_ratio/high_max": 0.031737884506583214, "clip_ratio/region_mean": 0.07950061978772283, "reward_total_mean": 0.8788509368896484, "reward_meter_mean": 0.9921952486038208, "reward_meter_std": 0.0020601593423634768, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9673808813095093, "reward_repeat_soft_std": 0.02818349003791809, "reward_judge_quality_mean": 0.6187499761581421, "reward_judge_quality_std": 0.24976776540279388, "reward_total_composite_mean": 0.8788509368896484, "reward_total_composite_std": 0.07591696828603745} {"timestamp_utc": "2026-04-13T04:28:55Z", "mode": "train", "global_step": 2446, "epoch": 0.2457056755399297, "loss": 0.0712, "grad_norm": 10.0440673828125, "learning_rate": 2.590909090909091e-06, "num_tokens": 4530119.0, "completions/mean_length": 86.0, "completions/min_length": 74.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.0, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.7018077373504639, "rewards/meter/std": 0.2318374365568161, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.979049801826477, "rewards/repeat_soft/std": 0.029202381148934364, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.78346848487854, "rewards/total_composite/std": 0.10685896873474121, "reward": 0.78346848487854, "reward_std": 0.10685897618532181, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14374369382858276, "sampling/sampling_logp_difference/max": 2.2303390502929688, "sampling/importance_sampling_ratio/min": 0.10749197751283646, "sampling/importance_sampling_ratio/mean": 1.0125652551651, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5386711917817593, "clip_ratio/low_mean": 0.04546409589238465, "clip_ratio/low_min": 0.04546409589238465, "clip_ratio/high_mean": 0.07468944136053324, "clip_ratio/high_max": 0.07468944136053324, "clip_ratio/region_mean": 0.12015353725291789, "reward_total_mean": 0.78346848487854, "reward_meter_mean": 0.7018077373504639, "reward_meter_std": 0.2318374365568161, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.979049801826477, "reward_repeat_soft_std": 0.029202381148934364, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.78346848487854, "reward_total_composite_std": 0.10685896873474121} {"timestamp_utc": "2026-04-13T04:29:02Z", "mode": "train", "global_step": 2447, "epoch": 0.24580612757408338, "loss": 0.0367, "grad_norm": 7.33669376373291, "learning_rate": 2.5878787878787883e-06, "num_tokens": 4532090.0, "completions/mean_length": 91.375, "completions/min_length": 81.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.375, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.4391997456550598, "rewards/meter/std": 0.34326687455177307, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9345816373825073, "rewards/repeat_soft/std": 0.08535746484994888, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.556035578250885, "rewards/total_composite/std": 0.140543594956398, "reward": 0.556035578250885, "reward_std": 0.140543594956398, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11191558092832565, "sampling/sampling_logp_difference/max": 2.8643107414245605, "sampling/importance_sampling_ratio/min": 0.057022418826818466, "sampling/importance_sampling_ratio/mean": 1.012633204460144, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.546048741787672, "clip_ratio/low_mean": 0.06081116572022438, "clip_ratio/low_min": 0.06081116572022438, "clip_ratio/high_mean": 0.0482941335067153, "clip_ratio/high_max": 0.0482941335067153, "clip_ratio/region_mean": 0.10910529922693968, "reward_total_mean": 0.556035578250885, "reward_meter_mean": 0.4391997456550598, "reward_meter_std": 0.34326687455177307, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9345816373825073, "reward_repeat_soft_std": 0.08535746484994888, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.556035578250885, "reward_total_composite_std": 0.140543594956398} {"timestamp_utc": "2026-04-13T04:29:09Z", "mode": "train", "global_step": 2448, "epoch": 0.24590657960823706, "loss": -0.0111, "grad_norm": 5.657322883605957, "learning_rate": 2.584848484848485e-06, "num_tokens": 4534734.0, "completions/mean_length": 138.5, "completions/min_length": 123.0, "completions/max_length": 151.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 138.5, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 151.0, "rewards/meter/mean": 0.9955586194992065, "rewards/meter/std": 0.001360128982923925, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.979133665561676, "rewards/repeat_soft/std": 0.011628672480583191, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.8202272653579712, "rewards/total_composite/std": 0.06896669417619705, "reward": 0.8202272653579712, "reward_std": 0.06896670162677765, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12109895795583725, "sampling/sampling_logp_difference/max": 1.618894100189209, "sampling/importance_sampling_ratio/min": 0.1981176733970642, "sampling/importance_sampling_ratio/mean": 1.004013180732727, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7118446454405785, "clip_ratio/low_mean": 0.04692574217915535, "clip_ratio/low_min": 0.04692574217915535, "clip_ratio/high_mean": 0.07795396819710732, "clip_ratio/high_max": 0.07795396819710732, "clip_ratio/region_mean": 0.12487971037626266, "reward_total_mean": 0.8202272653579712, "reward_meter_mean": 0.9955586194992065, "reward_meter_std": 0.001360128982923925, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.979133665561676, "reward_repeat_soft_std": 0.011628672480583191, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.8202272653579712, "reward_total_composite_std": 0.06896669417619705} {"timestamp_utc": "2026-04-13T04:29:15Z", "mode": "train", "global_step": 2449, "epoch": 0.24600703164239077, "loss": 0.0049, "grad_norm": 21.155935287475586, "learning_rate": 2.581818181818182e-06, "num_tokens": 4536345.0, "completions/mean_length": 28.375, "completions/min_length": 23.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.375, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.8811858892440796, "rewards/meter/std": 0.30699458718299866, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9554284811019897, "rewards/repeat_soft/std": 0.01420468557626009, "rewards/judge_quality/mean": 0.38374999165534973, "rewards/judge_quality/std": 0.11697832494974136, "rewards/total_composite/mean": 0.7572014927864075, "rewards/total_composite/std": 0.13930386304855347, "reward": 0.7572014927864075, "reward_std": 0.13930386304855347, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12395339459180832, "sampling/sampling_logp_difference/max": 1.4819226264953613, "sampling/importance_sampling_ratio/min": 0.227200448513031, "sampling/importance_sampling_ratio/mean": 0.9972270727157593, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.664886124432087, "clip_ratio/low_mean": 0.023747086990624666, "clip_ratio/low_min": 0.023747086990624666, "clip_ratio/high_mean": 0.09605529066175222, "clip_ratio/high_max": 0.09605529066175222, "clip_ratio/region_mean": 0.11980237765237689, "reward_total_mean": 0.7572014927864075, "reward_meter_mean": 0.8811858892440796, "reward_meter_std": 0.30699458718299866, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9554284811019897, "reward_repeat_soft_std": 0.01420468557626009, "reward_judge_quality_mean": 0.38374999165534973, "reward_judge_quality_std": 0.11697832494974136, "reward_total_composite_mean": 0.7572014927864075, "reward_total_composite_std": 0.13930386304855347} {"timestamp_utc": "2026-04-13T04:29:22Z", "mode": "train", "global_step": 2450, "epoch": 0.24610748367654445, "loss": -0.0109, "grad_norm": 11.097199440002441, "learning_rate": 2.5787878787878788e-06, "num_tokens": 4538833.0, "completions/mean_length": 115.0, "completions/min_length": 99.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.0, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.5045431852340698, "rewards/meter/std": 0.27350273728370667, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9427658319473267, "rewards/repeat_soft/std": 0.020936936140060425, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.6085709929466248, "rewards/total_composite/std": 0.12217266857624054, "reward": 0.6085709929466248, "reward_std": 0.12217266112565994, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12781278789043427, "sampling/sampling_logp_difference/max": 4.136970520019531, "sampling/importance_sampling_ratio/min": 0.015971163287758827, "sampling/importance_sampling_ratio/mean": 0.9944432377815247, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.418070774525404, "clip_ratio/low_mean": 0.035234310664236546, "clip_ratio/low_min": 0.035234310664236546, "clip_ratio/high_mean": 0.06045150803402066, "clip_ratio/high_max": 0.06045150803402066, "clip_ratio/region_mean": 0.09568581869825721, "reward_total_mean": 0.6085709929466248, "reward_meter_mean": 0.5045431852340698, "reward_meter_std": 0.27350273728370667, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9427658319473267, "reward_repeat_soft_std": 0.020936936140060425, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.6085709929466248, "reward_total_composite_std": 0.12217266857624054} {"timestamp_utc": "2026-04-13T04:30:32Z", "mode": "eval", "global_step": 2450, "epoch": 0.24610748367654445, "eval_loss": NaN, "eval_runtime": 69.8499, "eval_samples_per_second": 1.145, "eval_steps_per_second": 0.143, "eval_num_tokens": 4538833.0, "eval_completions/mean_length": 125.45, "eval_completions/min_length": 43.8, "eval_completions/max_length": 352.1, "eval_completions/clipped_ratio": 0.0625, "eval_completions/mean_terminated_length": 98.8125015258789, "eval_completions/min_terminated_length": 43.8, "eval_completions/max_terminated_length": 170.3, "eval_rewards/meter/mean": 0.8869106471538544, "eval_rewards/meter/std": 0.1779862177092582, "eval_rewards/count_adherence/mean": 0.9527083337306976, "eval_rewards/count_adherence/std": 0.09376556426286697, "eval_rewards/hard_gate/mean": 0.95, "eval_rewards/hard_gate/std": 0.1414213538169861, "eval_rewards/repeat_soft/mean": 0.9417082726955414, "eval_rewards/repeat_soft/std": 0.05172632858157158, "eval_rewards/judge_quality/mean": 0.3874999940395355, "eval_rewards/judge_quality/std": 0.1402278364636004, "eval_rewards/total_composite/mean": 0.7302121043205261, "eval_rewards/total_composite/std": 0.1711167089641094, "eval_reward": 0.7302121043205261, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.05231773518025875, "eval_sampling/sampling_logp_difference/max": 0.9563077926635742, "eval_sampling/importance_sampling_ratio/min": 0.3882982850074768, "eval_sampling/importance_sampling_ratio/mean": 1.0095563411712647, "eval_sampling/importance_sampling_ratio/max": 1.409497284889221, "eval_entropy": 0.5350359320640564, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7302121043205261, "eval_reward_meter_mean": 0.8869106471538544, "eval_reward_meter_std": 0.1779862177092582, "eval_reward_count_adherence_mean": 0.9527083337306976, "eval_reward_count_adherence_std": 0.09376556426286697, "eval_reward_hard_gate_mean": 0.95, "eval_reward_hard_gate_std": 0.1414213538169861, "eval_reward_repeat_soft_mean": 0.9417082726955414, "eval_reward_repeat_soft_std": 0.05172632858157158, "eval_reward_judge_quality_mean": 0.3874999940395355, "eval_reward_judge_quality_std": 0.1402278364636004, "eval_reward_total_composite_mean": 0.7302121043205261, "eval_reward_total_composite_std": 0.1711167089641094} {"timestamp_utc": "2026-04-13T04:30:43Z", "mode": "train", "global_step": 2451, "epoch": 0.24620793571069813, "loss": 0.0238, "grad_norm": 8.413737297058105, "learning_rate": 2.575757575757576e-06, "num_tokens": 4540699.0, "completions/mean_length": 75.25, "completions/min_length": 67.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.25, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.9866899251937866, "rewards/meter/std": 0.006193128880113363, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9334745407104492, "rewards/repeat_soft/std": 0.04846353828907013, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.12906256318092346, "rewards/total_composite/mean": 0.8193579316139221, "rewards/total_composite/std": 0.04040831699967384, "reward": 0.8193579316139221, "reward_std": 0.04040831699967384, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11144974827766418, "sampling/sampling_logp_difference/max": 1.8136191368103027, "sampling/importance_sampling_ratio/min": 0.1630629152059555, "sampling/importance_sampling_ratio/mean": 1.014086365699768, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.619136281311512, "clip_ratio/low_mean": 0.06546613259706646, "clip_ratio/low_min": 0.06546613259706646, "clip_ratio/high_mean": 0.014802631922066212, "clip_ratio/high_max": 0.014802631922066212, "clip_ratio/region_mean": 0.08026876451913267, "reward_total_mean": 0.8193579316139221, "reward_meter_mean": 0.9866899251937866, "reward_meter_std": 0.006193128880113363, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9334745407104492, "reward_repeat_soft_std": 0.04846353828907013, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.12906256318092346, "reward_total_composite_mean": 0.8193579316139221, "reward_total_composite_std": 0.04040831699967384} {"timestamp_utc": "2026-04-13T04:30:51Z", "mode": "train", "global_step": 2452, "epoch": 0.24630838774485184, "loss": 0.0784, "grad_norm": 9.45719051361084, "learning_rate": 2.572727272727273e-06, "num_tokens": 4542604.0, "completions/mean_length": 76.125, "completions/min_length": 63.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.125, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9684869050979614, "rewards/meter/std": 0.022398483008146286, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8150440454483032, "rewards/repeat_soft/std": 0.06734397262334824, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7933235168457031, "rewards/total_composite/std": 0.008037207648158073, "reward": 0.7933235168457031, "reward_std": 0.008037223480641842, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11011578887701035, "sampling/sampling_logp_difference/max": 1.8710412979125977, "sampling/importance_sampling_ratio/min": 0.15396325290203094, "sampling/importance_sampling_ratio/mean": 1.0082590579986572, "sampling/importance_sampling_ratio/max": 1.960314154624939, "entropy": 0.5799728408455849, "clip_ratio/low_mean": 0.028971029445528984, "clip_ratio/low_min": 0.028971029445528984, "clip_ratio/high_mean": 0.07552849990315735, "clip_ratio/high_max": 0.07552849990315735, "clip_ratio/region_mean": 0.10449952934868634, "reward_total_mean": 0.7933235168457031, "reward_meter_mean": 0.9684869050979614, "reward_meter_std": 0.022398483008146286, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8150440454483032, "reward_repeat_soft_std": 0.06734397262334824, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7933235168457031, "reward_total_composite_std": 0.008037207648158073} {"timestamp_utc": "2026-04-13T04:30:58Z", "mode": "train", "global_step": 2453, "epoch": 0.24640883977900552, "loss": 0.0208, "grad_norm": 6.824173927307129, "learning_rate": 2.5696969696969697e-06, "num_tokens": 4544281.0, "completions/mean_length": 58.625, "completions/min_length": 54.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.625, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9775043725967407, "rewards/meter/std": 0.011686488054692745, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9594907164573669, "rewards/repeat_soft/std": 0.033807411789894104, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8118259906768799, "rewards/total_composite/std": 0.0054605882614851, "reward": 0.8118259906768799, "reward_std": 0.005460601300001144, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08963152021169662, "sampling/sampling_logp_difference/max": 1.6380691528320312, "sampling/importance_sampling_ratio/min": 0.19435495138168335, "sampling/importance_sampling_ratio/mean": 1.0140827894210815, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4904583878815174, "clip_ratio/low_mean": 0.05184426298364997, "clip_ratio/low_min": 0.05184426298364997, "clip_ratio/high_mean": 0.0344768175855279, "clip_ratio/high_max": 0.0344768175855279, "clip_ratio/region_mean": 0.08632108056917787, "reward_total_mean": 0.8118259906768799, "reward_meter_mean": 0.9775043725967407, "reward_meter_std": 0.011686488054692745, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9594907164573669, "reward_repeat_soft_std": 0.033807411789894104, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8118259906768799, "reward_total_composite_std": 0.0054605882614851} {"timestamp_utc": "2026-04-13T04:31:10Z", "mode": "train", "global_step": 2454, "epoch": 0.24650929181315923, "loss": -0.1805, "grad_norm": 1.6252137422561646, "learning_rate": 2.566666666666667e-06, "num_tokens": 4546316.0, "completions/mean_length": 138.375, "completions/min_length": 76.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 85.0, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9316935539245605, "rewards/meter/std": 0.08765095472335815, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9075450897216797, "rewards/repeat_soft/std": 0.057642821222543716, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.18845234811306, "rewards/total_composite/mean": 0.7028694152832031, "rewards/total_composite/std": 0.2873639464378357, "reward": 0.7028694152832031, "reward_std": 0.2873639464378357, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10327617824077606, "sampling/sampling_logp_difference/max": 1.2325687408447266, "sampling/importance_sampling_ratio/min": 0.3015407919883728, "sampling/importance_sampling_ratio/mean": 1.0092483758926392, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4606349989771843, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09126879461109638, "clip_ratio/high_max": 0.09126879461109638, "clip_ratio/region_mean": 0.09126879461109638, "reward_total_mean": 0.7028694152832031, "reward_meter_mean": 0.9316935539245605, "reward_meter_std": 0.08765095472335815, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9075450897216797, "reward_repeat_soft_std": 0.057642821222543716, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.18845234811306, "reward_total_composite_mean": 0.7028694152832031, "reward_total_composite_std": 0.2873639464378357} {"timestamp_utc": "2026-04-13T04:31:21Z", "mode": "train", "global_step": 2455, "epoch": 0.2466097438473129, "loss": -0.112, "grad_norm": 2.4762587547302246, "learning_rate": 2.5636363636363638e-06, "num_tokens": 4547890.0, "completions/mean_length": 103.75, "completions/min_length": 43.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 45.42857360839844, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.7162813544273376, "rewards/meter/std": 0.4261735677719116, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9597644805908203, "rewards/repeat_soft/std": 0.02490968070924282, "rewards/judge_quality/mean": 0.3799999952316284, "rewards/judge_quality/std": 0.25707143545150757, "rewards/total_composite/mean": 0.6492834687232971, "rewards/total_composite/std": 0.3050076961517334, "reward": 0.6492834687232971, "reward_std": 0.305007666349411, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11405319720506668, "sampling/sampling_logp_difference/max": 1.39349365234375, "sampling/importance_sampling_ratio/min": 0.24820664525032043, "sampling/importance_sampling_ratio/mean": 1.0055755376815796, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5332050174474716, "clip_ratio/low_mean": 0.013888888992369175, "clip_ratio/low_min": 0.013888888992369175, "clip_ratio/high_mean": 0.0802554537076503, "clip_ratio/high_max": 0.0802554537076503, "clip_ratio/region_mean": 0.09414434270001948, "reward_total_mean": 0.6492834687232971, "reward_meter_mean": 0.7162813544273376, "reward_meter_std": 0.4261735677719116, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9597644805908203, "reward_repeat_soft_std": 0.02490968070924282, "reward_judge_quality_mean": 0.3799999952316284, "reward_judge_quality_std": 0.25707143545150757, "reward_total_composite_mean": 0.6492834687232971, "reward_total_composite_std": 0.3050076961517334} {"timestamp_utc": "2026-04-13T04:31:28Z", "mode": "train", "global_step": 2456, "epoch": 0.2467101958814666, "loss": 0.0234, "grad_norm": 10.877908706665039, "learning_rate": 2.5606060606060606e-06, "num_tokens": 4549411.0, "completions/mean_length": 43.125, "completions/min_length": 41.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.125, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.8751322031021118, "rewards/meter/std": 0.09070267528295517, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.96048903465271, "rewards/repeat_soft/std": 0.08757027238607407, "rewards/judge_quality/mean": 0.5600000023841858, "rewards/judge_quality/std": 0.22258226573467255, "rewards/total_composite/mean": 0.807858407497406, "rewards/total_composite/std": 0.08299248665571213, "reward": 0.807858407497406, "reward_std": 0.08299247920513153, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10340859740972519, "sampling/sampling_logp_difference/max": 1.2166681289672852, "sampling/importance_sampling_ratio/min": 0.3083251118659973, "sampling/importance_sampling_ratio/mean": 1.0236562490463257, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5774986557662487, "clip_ratio/low_mean": 0.07849755184724927, "clip_ratio/low_min": 0.07849755184724927, "clip_ratio/high_mean": 0.038690476678311825, "clip_ratio/high_max": 0.038690476678311825, "clip_ratio/region_mean": 0.1171880285255611, "reward_total_mean": 0.807858407497406, "reward_meter_mean": 0.8751322031021118, "reward_meter_std": 0.09070267528295517, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.96048903465271, "reward_repeat_soft_std": 0.08757027238607407, "reward_judge_quality_mean": 0.5600000023841858, "reward_judge_quality_std": 0.22258226573467255, "reward_total_composite_mean": 0.807858407497406, "reward_total_composite_std": 0.08299248665571213} {"timestamp_utc": "2026-04-13T04:31:34Z", "mode": "train", "global_step": 2457, "epoch": 0.2468106479156203, "loss": -0.0092, "grad_norm": 10.413339614868164, "learning_rate": 2.5575757575757574e-06, "num_tokens": 4551197.0, "completions/mean_length": 54.25, "completions/min_length": 48.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.25, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.7959468364715576, "rewards/meter/std": 0.2981356382369995, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9204997420310974, "rewards/repeat_soft/std": 0.08759219199419022, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.1865811049938202, "rewards/total_composite/mean": 0.7596009969711304, "rewards/total_composite/std": 0.1030246689915657, "reward": 0.7596009969711304, "reward_std": 0.10302466154098511, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10203731805086136, "sampling/sampling_logp_difference/max": 1.266822338104248, "sampling/importance_sampling_ratio/min": 0.2817254364490509, "sampling/importance_sampling_ratio/mean": 1.010118842124939, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5701357983052731, "clip_ratio/low_mean": 0.020096304826438427, "clip_ratio/low_min": 0.020096304826438427, "clip_ratio/high_mean": 0.07409946084953845, "clip_ratio/high_max": 0.07409946084953845, "clip_ratio/region_mean": 0.09419576567597687, "reward_total_mean": 0.7596009969711304, "reward_meter_mean": 0.7959468364715576, "reward_meter_std": 0.2981356382369995, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9204997420310974, "reward_repeat_soft_std": 0.08759219199419022, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.1865811049938202, "reward_total_composite_mean": 0.7596009969711304, "reward_total_composite_std": 0.1030246689915657} {"timestamp_utc": "2026-04-13T04:31:41Z", "mode": "train", "global_step": 2458, "epoch": 0.24691109994977398, "loss": 0.0421, "grad_norm": 12.27879810333252, "learning_rate": 2.5545454545454547e-06, "num_tokens": 4552821.0, "completions/mean_length": 48.0, "completions/min_length": 44.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.0, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9486411809921265, "rewards/meter/std": 0.02058618701994419, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9117552638053894, "rewards/repeat_soft/std": 0.093313068151474, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7974390387535095, "rewards/total_composite/std": 0.009295797906816006, "reward": 0.7974390387535095, "reward_std": 0.009295791387557983, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10525449365377426, "sampling/sampling_logp_difference/max": 3.1634716987609863, "sampling/importance_sampling_ratio/min": 0.0422787070274353, "sampling/importance_sampling_ratio/mean": 1.0077513456344604, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4916537515819073, "clip_ratio/low_mean": 0.04439295642077923, "clip_ratio/low_min": 0.04439295642077923, "clip_ratio/high_mean": 0.05174232926219702, "clip_ratio/high_max": 0.05174232926219702, "clip_ratio/region_mean": 0.09613528568297625, "reward_total_mean": 0.7974390387535095, "reward_meter_mean": 0.9486411809921265, "reward_meter_std": 0.02058618701994419, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9117552638053894, "reward_repeat_soft_std": 0.093313068151474, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7974390387535095, "reward_total_composite_std": 0.009295797906816006} {"timestamp_utc": "2026-04-13T04:31:47Z", "mode": "train", "global_step": 2459, "epoch": 0.24701155198392769, "loss": 0.0498, "grad_norm": 8.874593734741211, "learning_rate": 2.5515151515151515e-06, "num_tokens": 4554585.0, "completions/mean_length": 60.5, "completions/min_length": 50.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.5, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9716917276382446, "rewards/meter/std": 0.020265907049179077, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9504790902137756, "rewards/repeat_soft/std": 0.038462430238723755, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8139342069625854, "rewards/total_composite/std": 0.009448371827602386, "reward": 0.8139342069625854, "reward_std": 0.009448358789086342, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10515866428613663, "sampling/sampling_logp_difference/max": 2.280959367752075, "sampling/importance_sampling_ratio/min": 0.10218612104654312, "sampling/importance_sampling_ratio/mean": 1.0107223987579346, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5883234441280365, "clip_ratio/low_mean": 0.05985049158334732, "clip_ratio/low_min": 0.05985049158334732, "clip_ratio/high_mean": 0.06729246350005269, "clip_ratio/high_max": 0.06729246350005269, "clip_ratio/region_mean": 0.1271429550834, "reward_total_mean": 0.8139342069625854, "reward_meter_mean": 0.9716917276382446, "reward_meter_std": 0.020265907049179077, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9504790902137756, "reward_repeat_soft_std": 0.038462430238723755, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8139342069625854, "reward_total_composite_std": 0.009448371827602386} {"timestamp_utc": "2026-04-13T04:31:59Z", "mode": "train", "global_step": 2460, "epoch": 0.24711200401808137, "loss": -0.2207, "grad_norm": 1.6901178359985352, "learning_rate": 2.5484848484848484e-06, "num_tokens": 4556972.0, "completions/mean_length": 186.375, "completions/min_length": 130.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 139.85714721679688, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 148.0, "rewards/meter/mean": 0.8238378763198853, "rewards/meter/std": 0.2864617109298706, "rewards/count_adherence/mean": 0.925000011920929, "rewards/count_adherence/std": 0.2121320217847824, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.869978666305542, "rewards/repeat_soft/std": 0.13245859742164612, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.1345893144607544, "rewards/total_composite/mean": 0.659795343875885, "rewards/total_composite/std": 0.2715304493904114, "reward": 0.659795343875885, "reward_std": 0.2715304493904114, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09149402379989624, "sampling/sampling_logp_difference/max": 1.729259967803955, "sampling/importance_sampling_ratio/min": 0.19823291897773743, "sampling/importance_sampling_ratio/mean": 1.0101962089538574, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4834476634860039, "clip_ratio/low_mean": 0.008992806077003479, "clip_ratio/low_min": 0.008992806077003479, "clip_ratio/high_mean": 0.057854092214256525, "clip_ratio/high_max": 0.057854092214256525, "clip_ratio/region_mean": 0.06684689829126, "reward_total_mean": 0.659795343875885, "reward_meter_mean": 0.8238378763198853, "reward_meter_std": 0.2864617109298706, "reward_count_adherence_mean": 0.925000011920929, "reward_count_adherence_std": 0.2121320217847824, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.869978666305542, "reward_repeat_soft_std": 0.13245859742164612, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.1345893144607544, "reward_total_composite_mean": 0.659795343875885, "reward_total_composite_std": 0.2715304493904114} {"timestamp_utc": "2026-04-13T04:32:06Z", "mode": "train", "global_step": 2461, "epoch": 0.24721245605223505, "loss": 0.0936, "grad_norm": 11.038715362548828, "learning_rate": 2.5454545454545456e-06, "num_tokens": 4558672.0, "completions/mean_length": 57.5, "completions/min_length": 50.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.5, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.8402984142303467, "rewards/meter/std": 0.29152384400367737, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9640182256698608, "rewards/repeat_soft/std": 0.03084646724164486, "rewards/judge_quality/mean": 0.5024999976158142, "rewards/judge_quality/std": 0.21224987506866455, "rewards/total_composite/mean": 0.775286078453064, "rewards/total_composite/std": 0.15888568758964539, "reward": 0.775286078453064, "reward_std": 0.158885657787323, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11932864040136337, "sampling/sampling_logp_difference/max": 2.5231399536132812, "sampling/importance_sampling_ratio/min": 0.08020736277103424, "sampling/importance_sampling_ratio/mean": 0.997890055179596, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6376130469143391, "clip_ratio/low_mean": 0.030390458647161722, "clip_ratio/low_min": 0.030390458647161722, "clip_ratio/high_mean": 0.08211624901741743, "clip_ratio/high_max": 0.08211624901741743, "clip_ratio/region_mean": 0.11250670766457915, "reward_total_mean": 0.775286078453064, "reward_meter_mean": 0.8402984142303467, "reward_meter_std": 0.29152384400367737, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9640182256698608, "reward_repeat_soft_std": 0.03084646724164486, "reward_judge_quality_mean": 0.5024999976158142, "reward_judge_quality_std": 0.21224987506866455, "reward_total_composite_mean": 0.775286078453064, "reward_total_composite_std": 0.15888568758964539} {"timestamp_utc": "2026-04-13T04:32:13Z", "mode": "train", "global_step": 2462, "epoch": 0.24731290808638876, "loss": 0.0769, "grad_norm": 14.086212158203125, "learning_rate": 2.542424242424243e-06, "num_tokens": 4560385.0, "completions/mean_length": 55.125, "completions/min_length": 49.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.125, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.8335235118865967, "rewards/meter/std": 0.3365897536277771, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9805172681808472, "rewards/repeat_soft/std": 0.02739836275577545, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.7472622990608215, "rewards/total_composite/std": 0.1558343917131424, "reward": 0.7472622990608215, "reward_std": 0.1558343768119812, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09914567321538925, "sampling/sampling_logp_difference/max": 1.2584681510925293, "sampling/importance_sampling_ratio/min": 0.2840888798236847, "sampling/importance_sampling_ratio/mean": 0.9902831315994263, "sampling/importance_sampling_ratio/max": 1.6258456707000732, "entropy": 0.5906777828931808, "clip_ratio/low_mean": 0.025423728860914707, "clip_ratio/low_min": 0.025423728860914707, "clip_ratio/high_mean": 0.07196213584393263, "clip_ratio/high_max": 0.07196213584393263, "clip_ratio/region_mean": 0.09738586470484734, "reward_total_mean": 0.7472622990608215, "reward_meter_mean": 0.8335235118865967, "reward_meter_std": 0.3365897536277771, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9805172681808472, "reward_repeat_soft_std": 0.02739836275577545, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.7472622990608215, "reward_total_composite_std": 0.1558343917131424} {"timestamp_utc": "2026-04-13T04:32:21Z", "mode": "train", "global_step": 2463, "epoch": 0.24741336012054244, "loss": 0.0246, "grad_norm": 4.78108549118042, "learning_rate": 2.5393939393939397e-06, "num_tokens": 4562867.0, "completions/mean_length": 139.25, "completions/min_length": 133.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 139.25, "completions/min_terminated_length": 133.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.9944088459014893, "rewards/meter/std": 0.0018450012430548668, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9674877524375916, "rewards/repeat_soft/std": 0.011956961825489998, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.25150617957115173, "rewards/total_composite/mean": 0.8513576984405518, "rewards/total_composite/std": 0.07556350529193878, "reward": 0.8513576984405518, "reward_std": 0.07556349784135818, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09294813871383667, "sampling/sampling_logp_difference/max": 1.4384593963623047, "sampling/importance_sampling_ratio/min": 0.2372930496931076, "sampling/importance_sampling_ratio/mean": 1.0042259693145752, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6034104973077774, "clip_ratio/low_mean": 0.06805644277483225, "clip_ratio/low_min": 0.06805644277483225, "clip_ratio/high_mean": 0.02507508359849453, "clip_ratio/high_max": 0.02507508359849453, "clip_ratio/region_mean": 0.09313152637332678, "reward_total_mean": 0.8513576984405518, "reward_meter_mean": 0.9944088459014893, "reward_meter_std": 0.0018450012430548668, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9674877524375916, "reward_repeat_soft_std": 0.011956961825489998, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.25150617957115173, "reward_total_composite_mean": 0.8513576984405518, "reward_total_composite_std": 0.07556350529193878} {"timestamp_utc": "2026-04-13T04:32:29Z", "mode": "train", "global_step": 2464, "epoch": 0.24751381215469614, "loss": 0.0016, "grad_norm": 6.476612567901611, "learning_rate": 2.536363636363637e-06, "num_tokens": 4565337.0, "completions/mean_length": 138.75, "completions/min_length": 117.0, "completions/max_length": 149.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 138.75, "completions/min_terminated_length": 117.0, "completions/max_terminated_length": 149.0, "rewards/meter/mean": 0.8385910987854004, "rewards/meter/std": 0.2252875119447708, "rewards/count_adherence/mean": 0.8500000238418579, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8669912815093994, "rewards/repeat_soft/std": 0.05767088010907173, "rewards/judge_quality/mean": 0.3474999964237213, "rewards/judge_quality/std": 0.11310549825429916, "rewards/total_composite/mean": 0.6958151459693909, "rewards/total_composite/std": 0.09738092124462128, "reward": 0.6958151459693909, "reward_std": 0.09738090634346008, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0876745954155922, "sampling/sampling_logp_difference/max": 1.4557275772094727, "sampling/importance_sampling_ratio/min": 0.2332306057214737, "sampling/importance_sampling_ratio/mean": 1.0076605081558228, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5040974617004395, "clip_ratio/low_mean": 0.022653449326753616, "clip_ratio/low_min": 0.022653449326753616, "clip_ratio/high_mean": 0.05800102837383747, "clip_ratio/high_max": 0.05800102837383747, "clip_ratio/region_mean": 0.08065447770059109, "reward_total_mean": 0.6958151459693909, "reward_meter_mean": 0.8385910987854004, "reward_meter_std": 0.2252875119447708, "reward_count_adherence_mean": 0.8500000238418579, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8669912815093994, "reward_repeat_soft_std": 0.05767088010907173, "reward_judge_quality_mean": 0.3474999964237213, "reward_judge_quality_std": 0.11310549825429916, "reward_total_composite_mean": 0.6958151459693909, "reward_total_composite_std": 0.09738092124462128} {"timestamp_utc": "2026-04-13T04:32:40Z", "mode": "train", "global_step": 2465, "epoch": 0.24761426418884983, "loss": -0.0872, "grad_norm": 5.511695861816406, "learning_rate": 2.5333333333333338e-06, "num_tokens": 4567030.0, "completions/mean_length": 112.625, "completions/min_length": 52.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 55.57143020629883, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.7619497776031494, "rewards/meter/std": 0.39210742712020874, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9709028005599976, "rewards/repeat_soft/std": 0.029941588640213013, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.13617216050624847, "rewards/total_composite/mean": 0.5695562362670898, "rewards/total_composite/std": 0.3699016571044922, "reward": 0.5695562362670898, "reward_std": 0.3699016273021698, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13959765434265137, "sampling/sampling_logp_difference/max": 2.9238696098327637, "sampling/importance_sampling_ratio/min": 0.053725387901067734, "sampling/importance_sampling_ratio/mean": 1.00938880443573, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6129428967833519, "clip_ratio/low_mean": 0.02552447561174631, "clip_ratio/low_min": 0.02552447561174631, "clip_ratio/high_mean": 0.06250071432441473, "clip_ratio/high_max": 0.06250071432441473, "clip_ratio/region_mean": 0.08802518993616104, "reward_total_mean": 0.5695562362670898, "reward_meter_mean": 0.7619497776031494, "reward_meter_std": 0.39210742712020874, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9709028005599976, "reward_repeat_soft_std": 0.029941588640213013, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.13617216050624847, "reward_total_composite_mean": 0.5695562362670898, "reward_total_composite_std": 0.3699016571044922} {"timestamp_utc": "2026-04-13T04:32:48Z", "mode": "train", "global_step": 2466, "epoch": 0.2477147162230035, "loss": 0.0042, "grad_norm": 4.223349094390869, "learning_rate": 2.5303030303030306e-06, "num_tokens": 4569793.0, "completions/mean_length": 142.375, "completions/min_length": 121.0, "completions/max_length": 163.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.375, "completions/min_terminated_length": 121.0, "completions/max_terminated_length": 163.0, "rewards/meter/mean": 0.9875025749206543, "rewards/meter/std": 0.004699964541941881, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8432137966156006, "rewards/repeat_soft/std": 0.10237132012844086, "rewards/judge_quality/mean": 0.2887499928474426, "rewards/judge_quality/std": 0.11630470305681229, "rewards/total_composite/mean": 0.7559475302696228, "rewards/total_composite/std": 0.038829464465379715, "reward": 0.7559475302696228, "reward_std": 0.038829464465379715, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08519516885280609, "sampling/sampling_logp_difference/max": 2.019801378250122, "sampling/importance_sampling_ratio/min": 0.13268181681632996, "sampling/importance_sampling_ratio/mean": 1.0017210245132446, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4787338636815548, "clip_ratio/low_mean": 0.028855863958597183, "clip_ratio/low_min": 0.028855863958597183, "clip_ratio/high_mean": 0.04377408139407635, "clip_ratio/high_max": 0.04377408139407635, "clip_ratio/region_mean": 0.07262994535267353, "reward_total_mean": 0.7559475302696228, "reward_meter_mean": 0.9875025749206543, "reward_meter_std": 0.004699964541941881, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8432137966156006, "reward_repeat_soft_std": 0.10237132012844086, "reward_judge_quality_mean": 0.2887499928474426, "reward_judge_quality_std": 0.11630470305681229, "reward_total_composite_mean": 0.7559475302696228, "reward_total_composite_std": 0.038829464465379715} {"timestamp_utc": "2026-04-13T04:32:54Z", "mode": "train", "global_step": 2467, "epoch": 0.24781516825715721, "loss": 0.05, "grad_norm": 15.31036376953125, "learning_rate": 2.5272727272727274e-06, "num_tokens": 4571220.0, "completions/mean_length": 29.375, "completions/min_length": 28.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.375, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.806338906288147, "rewards/meter/std": 0.3404441177845001, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9272210001945496, "rewards/repeat_soft/std": 0.030679944902658463, "rewards/judge_quality/mean": 0.5012500286102295, "rewards/judge_quality/std": 0.16974246501922607, "rewards/total_composite/mean": 0.7559496164321899, "rewards/total_composite/std": 0.11798973381519318, "reward": 0.7559496164321899, "reward_std": 0.11798974126577377, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10760976374149323, "sampling/sampling_logp_difference/max": 1.5572218894958496, "sampling/importance_sampling_ratio/min": 0.21072065830230713, "sampling/importance_sampling_ratio/mean": 1.0128103494644165, "sampling/importance_sampling_ratio/max": 1.5833708047866821, "entropy": 0.6933459565043449, "clip_ratio/low_mean": 0.039015152491629124, "clip_ratio/low_min": 0.039015152491629124, "clip_ratio/high_mean": 0.06142241507768631, "clip_ratio/high_max": 0.06142241507768631, "clip_ratio/region_mean": 0.10043756756931543, "reward_total_mean": 0.7559496164321899, "reward_meter_mean": 0.806338906288147, "reward_meter_std": 0.3404441177845001, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9272210001945496, "reward_repeat_soft_std": 0.030679944902658463, "reward_judge_quality_mean": 0.5012500286102295, "reward_judge_quality_std": 0.16974246501922607, "reward_total_composite_mean": 0.7559496164321899, "reward_total_composite_std": 0.11798973381519318} {"timestamp_utc": "2026-04-13T04:33:06Z", "mode": "train", "global_step": 2468, "epoch": 0.2479156202913109, "loss": -0.1468, "grad_norm": 1.972063422203064, "learning_rate": 2.5242424242424247e-06, "num_tokens": 4572923.0, "completions/mean_length": 113.875, "completions/min_length": 52.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 57.000003814697266, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.8244485259056091, "rewards/meter/std": 0.3380776047706604, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9881374835968018, "rewards/repeat_soft/std": 0.0214268546551466, "rewards/judge_quality/mean": 0.3712499737739563, "rewards/judge_quality/std": 0.1470119059085846, "rewards/total_composite/mean": 0.6960833072662354, "rewards/total_composite/std": 0.2860865294933319, "reward": 0.6960833072662354, "reward_std": 0.2860864996910095, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13837830722332, "sampling/sampling_logp_difference/max": 1.925682544708252, "sampling/importance_sampling_ratio/min": 0.14577622711658478, "sampling/importance_sampling_ratio/mean": 1.0019999742507935, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6290586218237877, "clip_ratio/low_mean": 0.023148147389292717, "clip_ratio/low_min": 0.023148147389292717, "clip_ratio/high_mean": 0.10884390771389008, "clip_ratio/high_max": 0.10884390771389008, "clip_ratio/region_mean": 0.1319920551031828, "reward_total_mean": 0.6960833072662354, "reward_meter_mean": 0.8244485259056091, "reward_meter_std": 0.3380776047706604, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9881374835968018, "reward_repeat_soft_std": 0.0214268546551466, "reward_judge_quality_mean": 0.3712499737739563, "reward_judge_quality_std": 0.1470119059085846, "reward_total_composite_mean": 0.6960833072662354, "reward_total_composite_std": 0.2860865294933319} {"timestamp_utc": "2026-04-13T04:33:13Z", "mode": "train", "global_step": 2469, "epoch": 0.2480160723254646, "loss": 0.0278, "grad_norm": 16.023845672607422, "learning_rate": 2.5212121212121215e-06, "num_tokens": 4574351.0, "completions/mean_length": 28.5, "completions/min_length": 23.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.5, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.7356624603271484, "rewards/meter/std": 0.40646642446517944, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9118325710296631, "rewards/repeat_soft/std": 0.08139326423406601, "rewards/judge_quality/mean": 0.3799999952316284, "rewards/judge_quality/std": 0.11501552164554596, "rewards/total_composite/mean": 0.6862313151359558, "rewards/total_composite/std": 0.18431566655635834, "reward": 0.6862313151359558, "reward_std": 0.18431566655635834, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11927258968353271, "sampling/sampling_logp_difference/max": 1.5663087368011475, "sampling/importance_sampling_ratio/min": 0.20881454646587372, "sampling/importance_sampling_ratio/mean": 1.030469298362732, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9875640347599983, "clip_ratio/low_mean": 0.051070602145045996, "clip_ratio/low_min": 0.051070602145045996, "clip_ratio/high_mean": 0.06892328057438135, "clip_ratio/high_max": 0.06892328057438135, "clip_ratio/region_mean": 0.11999388271942735, "reward_total_mean": 0.6862313151359558, "reward_meter_mean": 0.7356624603271484, "reward_meter_std": 0.40646642446517944, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9118325710296631, "reward_repeat_soft_std": 0.08139326423406601, "reward_judge_quality_mean": 0.3799999952316284, "reward_judge_quality_std": 0.11501552164554596, "reward_total_composite_mean": 0.6862313151359558, "reward_total_composite_std": 0.18431566655635834} {"timestamp_utc": "2026-04-13T04:33:21Z", "mode": "train", "global_step": 2470, "epoch": 0.24811652435961828, "loss": 0.0709, "grad_norm": 9.570782661437988, "learning_rate": 2.5181818181818184e-06, "num_tokens": 4576089.0, "completions/mean_length": 48.25, "completions/min_length": 42.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.25, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.9788758754730225, "rewards/meter/std": 0.014274325221776962, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9591895341873169, "rewards/repeat_soft/std": 0.029048683121800423, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404937028885, "rewards/total_composite/mean": 0.8532880544662476, "rewards/total_composite/std": 0.06692435592412949, "reward": 0.8532880544662476, "reward_std": 0.06692434847354889, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12126588076353073, "sampling/sampling_logp_difference/max": 2.216435432434082, "sampling/importance_sampling_ratio/min": 0.10899695008993149, "sampling/importance_sampling_ratio/mean": 1.0177689790725708, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.696501687169075, "clip_ratio/low_mean": 0.0761182731948793, "clip_ratio/low_min": 0.0761182731948793, "clip_ratio/high_mean": 0.025202634744346142, "clip_ratio/high_max": 0.025202634744346142, "clip_ratio/region_mean": 0.10132090793922544, "reward_total_mean": 0.8532880544662476, "reward_meter_mean": 0.9788758754730225, "reward_meter_std": 0.014274325221776962, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9591895341873169, "reward_repeat_soft_std": 0.029048683121800423, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404937028885, "reward_total_composite_mean": 0.8532880544662476, "reward_total_composite_std": 0.06692435592412949} {"timestamp_utc": "2026-04-13T04:33:28Z", "mode": "train", "global_step": 2471, "epoch": 0.24821697639377197, "loss": 0.029, "grad_norm": 12.157599449157715, "learning_rate": 2.5151515151515156e-06, "num_tokens": 4577592.0, "completions/mean_length": 51.875, "completions/min_length": 45.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.875, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.8504254817962646, "rewards/meter/std": 0.29338401556015015, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9605950117111206, "rewards/repeat_soft/std": 0.03146599605679512, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.7757509350776672, "rewards/total_composite/std": 0.15054844319820404, "reward": 0.7757509350776672, "reward_std": 0.15054844319820404, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12088479846715927, "sampling/sampling_logp_difference/max": 1.5836734771728516, "sampling/importance_sampling_ratio/min": 0.2052198350429535, "sampling/importance_sampling_ratio/mean": 1.0074344873428345, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7426760718226433, "clip_ratio/low_mean": 0.03479588497430086, "clip_ratio/low_min": 0.03479588497430086, "clip_ratio/high_mean": 0.08577895909547806, "clip_ratio/high_max": 0.08577895909547806, "clip_ratio/region_mean": 0.12057484406977892, "reward_total_mean": 0.7757509350776672, "reward_meter_mean": 0.8504254817962646, "reward_meter_std": 0.29338401556015015, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9605950117111206, "reward_repeat_soft_std": 0.03146599605679512, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.7757509350776672, "reward_total_composite_std": 0.15054844319820404} {"timestamp_utc": "2026-04-13T04:33:40Z", "mode": "train", "global_step": 2472, "epoch": 0.24831742842792567, "loss": -0.1541, "grad_norm": 2.657001256942749, "learning_rate": 2.5121212121212125e-06, "num_tokens": 4579995.0, "completions/mean_length": 181.375, "completions/min_length": 119.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 134.1428680419922, "completions/min_terminated_length": 119.0, "completions/max_terminated_length": 144.0, "rewards/meter/mean": 0.9570776224136353, "rewards/meter/std": 0.02871902473270893, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8170425891876221, "rewards/repeat_soft/std": 0.0930669978260994, "rewards/judge_quality/mean": 0.2762500047683716, "rewards/judge_quality/std": 0.13689802587032318, "rewards/total_composite/mean": 0.7190141677856445, "rewards/total_composite/std": 0.05219382420182228, "reward": 0.7190141677856445, "reward_std": 0.05219383165240288, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07518467307090759, "sampling/sampling_logp_difference/max": 1.3880001306533813, "sampling/importance_sampling_ratio/min": 0.24957391619682312, "sampling/importance_sampling_ratio/mean": 1.0071120262145996, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3164690174162388, "clip_ratio/low_mean": 0.0180124226026237, "clip_ratio/low_min": 0.0180124226026237, "clip_ratio/high_mean": 0.05141937220469117, "clip_ratio/high_max": 0.05141937220469117, "clip_ratio/region_mean": 0.06943179480731487, "reward_total_mean": 0.7190141677856445, "reward_meter_mean": 0.9570776224136353, "reward_meter_std": 0.02871902473270893, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8170425891876221, "reward_repeat_soft_std": 0.0930669978260994, "reward_judge_quality_mean": 0.2762500047683716, "reward_judge_quality_std": 0.13689802587032318, "reward_total_composite_mean": 0.7190141677856445, "reward_total_composite_std": 0.05219382420182228} {"timestamp_utc": "2026-04-13T04:33:53Z", "mode": "train", "global_step": 2473, "epoch": 0.24841788046207935, "loss": -0.1971, "grad_norm": 2.3726089000701904, "learning_rate": 2.5090909090909093e-06, "num_tokens": 4582688.0, "completions/mean_length": 210.625, "completions/min_length": 150.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 167.57144165039062, "completions/min_terminated_length": 150.0, "completions/max_terminated_length": 178.0, "rewards/meter/mean": 0.7747876644134521, "rewards/meter/std": 0.32546117901802063, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.1414213478565216, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9488843679428101, "rewards/repeat_soft/std": 0.03460919111967087, "rewards/judge_quality/mean": 0.2549999952316284, "rewards/judge_quality/std": 0.12398156523704529, "rewards/total_composite/mean": 0.5820004940032959, "rewards/total_composite/std": 0.2704005241394043, "reward": 0.5820004940032959, "reward_std": 0.2704004943370819, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10636227577924728, "sampling/sampling_logp_difference/max": 1.581695556640625, "sampling/importance_sampling_ratio/min": 0.20562615990638733, "sampling/importance_sampling_ratio/mean": 1.0094459056854248, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5706547051668167, "clip_ratio/low_mean": 0.0239863945171237, "clip_ratio/low_min": 0.0239863945171237, "clip_ratio/high_mean": 0.06018059328198433, "clip_ratio/high_max": 0.06018059328198433, "clip_ratio/region_mean": 0.08416698779910803, "reward_total_mean": 0.5820004940032959, "reward_meter_mean": 0.7747876644134521, "reward_meter_std": 0.32546117901802063, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.1414213478565216, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9488843679428101, "reward_repeat_soft_std": 0.03460919111967087, "reward_judge_quality_mean": 0.2549999952316284, "reward_judge_quality_std": 0.12398156523704529, "reward_total_composite_mean": 0.5820004940032959, "reward_total_composite_std": 0.2704005241394043} {"timestamp_utc": "2026-04-13T04:34:00Z", "mode": "train", "global_step": 2474, "epoch": 0.24851833249623304, "loss": 0.0287, "grad_norm": 8.746886253356934, "learning_rate": 2.506060606060606e-06, "num_tokens": 4584471.0, "completions/mean_length": 62.875, "completions/min_length": 54.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.875, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9777086973190308, "rewards/meter/std": 0.02477523498237133, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9690234661102295, "rewards/repeat_soft/std": 0.027188558131456375, "rewards/judge_quality/mean": 0.42124998569488525, "rewards/judge_quality/std": 0.06998724490404129, "rewards/total_composite/mean": 0.8132462501525879, "rewards/total_composite/std": 0.02039412036538124, "reward": 0.8132462501525879, "reward_std": 0.02039412036538124, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1191539540886879, "sampling/sampling_logp_difference/max": 1.5505692958831787, "sampling/importance_sampling_ratio/min": 0.2121271789073944, "sampling/importance_sampling_ratio/mean": 1.0205273628234863, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7780640721321106, "clip_ratio/low_mean": 0.04177634138613939, "clip_ratio/low_min": 0.04177634138613939, "clip_ratio/high_mean": 0.08269652538001537, "clip_ratio/high_max": 0.08269652538001537, "clip_ratio/region_mean": 0.12447286676615477, "reward_total_mean": 0.8132462501525879, "reward_meter_mean": 0.9777086973190308, "reward_meter_std": 0.02477523498237133, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9690234661102295, "reward_repeat_soft_std": 0.027188558131456375, "reward_judge_quality_mean": 0.42124998569488525, "reward_judge_quality_std": 0.06998724490404129, "reward_total_composite_mean": 0.8132462501525879, "reward_total_composite_std": 0.02039412036538124} {"timestamp_utc": "2026-04-13T04:34:08Z", "mode": "train", "global_step": 2475, "epoch": 0.24861878453038674, "loss": 0.041, "grad_norm": 14.18317985534668, "learning_rate": 2.5030303030303034e-06, "num_tokens": 4585818.0, "completions/mean_length": 26.375, "completions/min_length": 22.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.375, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.9863632917404175, "rewards/meter/std": 0.012091665528714657, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.952028751373291, "rewards/repeat_soft/std": 0.020283527672290802, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.890066385269165, "rewards/total_composite/std": 0.07831361889839172, "reward": 0.890066385269165, "reward_std": 0.07831361889839172, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11268766224384308, "sampling/sampling_logp_difference/max": 1.4855690002441406, "sampling/importance_sampling_ratio/min": 0.22637349367141724, "sampling/importance_sampling_ratio/mean": 0.9982537627220154, "sampling/importance_sampling_ratio/max": 1.7640572786331177, "entropy": 0.6139375604689121, "clip_ratio/low_mean": 0.042066147085279226, "clip_ratio/low_min": 0.042066147085279226, "clip_ratio/high_mean": 0.06422078050673008, "clip_ratio/high_max": 0.06422078050673008, "clip_ratio/region_mean": 0.1062869275920093, "reward_total_mean": 0.890066385269165, "reward_meter_mean": 0.9863632917404175, "reward_meter_std": 0.012091665528714657, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.952028751373291, "reward_repeat_soft_std": 0.020283527672290802, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.890066385269165, "reward_total_composite_std": 0.07831361889839172} {"timestamp_utc": "2026-04-13T04:34:16Z", "mode": "train", "global_step": 2476, "epoch": 0.24871923656454042, "loss": 0.0347, "grad_norm": 19.104602813720703, "learning_rate": 2.5e-06, "num_tokens": 4587247.0, "completions/mean_length": 22.625, "completions/min_length": 21.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.625, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.9037047624588013, "rewards/meter/std": 0.07212629169225693, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.766167163848877, "rewards/total_composite/std": 0.03148523345589638, "reward": 0.766167163848877, "reward_std": 0.03148524463176727, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13279151916503906, "sampling/sampling_logp_difference/max": 1.1038408279418945, "sampling/importance_sampling_ratio/min": 0.3315950334072113, "sampling/importance_sampling_ratio/mean": 1.0280295610427856, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7206790596246719, "clip_ratio/low_mean": 0.0992911271750927, "clip_ratio/low_min": 0.0992911271750927, "clip_ratio/high_mean": 0.049948240630328655, "clip_ratio/high_max": 0.049948240630328655, "clip_ratio/region_mean": 0.14923936780542135, "reward_total_mean": 0.766167163848877, "reward_meter_mean": 0.9037047624588013, "reward_meter_std": 0.07212629169225693, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.766167163848877, "reward_total_composite_std": 0.03148523345589638} {"timestamp_utc": "2026-04-13T04:34:30Z", "mode": "train", "global_step": 2477, "epoch": 0.24881968859869413, "loss": -0.1514, "grad_norm": 1.7959699630737305, "learning_rate": 2.496969696969697e-06, "num_tokens": 4589015.0, "completions/mean_length": 245.0, "completions/min_length": 79.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 84.80000305175781, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.7370200157165527, "rewards/meter/std": 0.39014968276023865, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.24800792336463928, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9858424067497253, "rewards/repeat_soft/std": 0.01659271866083145, "rewards/judge_quality/mean": 0.2475000023841858, "rewards/judge_quality/std": 0.18729273974895477, "rewards/total_composite/mean": 0.49860888719558716, "rewards/total_composite/std": 0.4140736162662506, "reward": 0.49860888719558716, "reward_std": 0.41407355666160583, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13785845041275024, "sampling/sampling_logp_difference/max": 1.9742207527160645, "sampling/importance_sampling_ratio/min": 0.1388694792985916, "sampling/importance_sampling_ratio/mean": 1.014304757118225, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5329064354300499, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.07833083625882864, "clip_ratio/high_max": 0.07833083625882864, "clip_ratio/region_mean": 0.07833083625882864, "reward_total_mean": 0.49860888719558716, "reward_meter_mean": 0.7370200157165527, "reward_meter_std": 0.39014968276023865, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.24800792336463928, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9858424067497253, "reward_repeat_soft_std": 0.01659271866083145, "reward_judge_quality_mean": 0.2475000023841858, "reward_judge_quality_std": 0.18729273974895477, "reward_total_composite_mean": 0.49860888719558716, "reward_total_composite_std": 0.4140736162662506} {"timestamp_utc": "2026-04-13T04:34:42Z", "mode": "train", "global_step": 2478, "epoch": 0.24892014063284781, "loss": -0.1384, "grad_norm": 1.5607656240463257, "learning_rate": 2.4939393939393943e-06, "num_tokens": 4590688.0, "completions/mean_length": 112.125, "completions/min_length": 47.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 55.000003814697266, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9185711145401001, "rewards/meter/std": 0.1529351770877838, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9920239448547363, "rewards/repeat_soft/std": 0.010144942440092564, "rewards/judge_quality/mean": 0.5724999904632568, "rewards/judge_quality/std": 0.31702864170074463, "rewards/total_composite/mean": 0.770452618598938, "rewards/total_composite/std": 0.32108718156814575, "reward": 0.770452618598938, "reward_std": 0.32108718156814575, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09606321156024933, "sampling/sampling_logp_difference/max": 1.2310876846313477, "sampling/importance_sampling_ratio/min": 0.291974812746048, "sampling/importance_sampling_ratio/mean": 0.9972899556159973, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4854641817510128, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10004722978919744, "clip_ratio/high_max": 0.10004722978919744, "clip_ratio/region_mean": 0.10004722978919744, "reward_total_mean": 0.770452618598938, "reward_meter_mean": 0.9185711145401001, "reward_meter_std": 0.1529351770877838, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9920239448547363, "reward_repeat_soft_std": 0.010144942440092564, "reward_judge_quality_mean": 0.5724999904632568, "reward_judge_quality_std": 0.31702864170074463, "reward_total_composite_mean": 0.770452618598938, "reward_total_composite_std": 0.32108718156814575} {"timestamp_utc": "2026-04-13T04:34:49Z", "mode": "train", "global_step": 2479, "epoch": 0.2490205926670015, "loss": 0.0115, "grad_norm": 7.21622896194458, "learning_rate": 2.490909090909091e-06, "num_tokens": 4592990.0, "completions/mean_length": 112.75, "completions/min_length": 108.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.75, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.9387017488479614, "rewards/meter/std": 0.10664942115545273, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9317355751991272, "rewards/repeat_soft/std": 0.03778616711497307, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7915893197059631, "rewards/total_composite/std": 0.04625541344285011, "reward": 0.7915893197059631, "reward_std": 0.04625542089343071, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09845834225416183, "sampling/sampling_logp_difference/max": 2.5816922187805176, "sampling/importance_sampling_ratio/min": 0.07564588636159897, "sampling/importance_sampling_ratio/mean": 1.012061357498169, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6101342923939228, "clip_ratio/low_mean": 0.009009009227156639, "clip_ratio/low_min": 0.009009009227156639, "clip_ratio/high_mean": 0.08182366844266653, "clip_ratio/high_max": 0.08182366844266653, "clip_ratio/region_mean": 0.09083267766982317, "reward_total_mean": 0.7915893197059631, "reward_meter_mean": 0.9387017488479614, "reward_meter_std": 0.10664942115545273, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9317355751991272, "reward_repeat_soft_std": 0.03778616711497307, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7915893197059631, "reward_total_composite_std": 0.04625541344285011} {"timestamp_utc": "2026-04-13T04:34:57Z", "mode": "train", "global_step": 2480, "epoch": 0.2491210447011552, "loss": 0.0326, "grad_norm": 7.209420680999756, "learning_rate": 2.487878787878788e-06, "num_tokens": 4595084.0, "completions/mean_length": 87.75, "completions/min_length": 79.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.75, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.8117817640304565, "rewards/meter/std": 0.29429498314857483, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9583067893981934, "rewards/repeat_soft/std": 0.023175112903118134, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.7142574787139893, "rewards/total_composite/std": 0.12467237561941147, "reward": 0.7142574787139893, "reward_std": 0.12467236816883087, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10250992327928543, "sampling/sampling_logp_difference/max": 1.464754581451416, "sampling/importance_sampling_ratio/min": 0.23113471269607544, "sampling/importance_sampling_ratio/mean": 1.0082600116729736, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7705654129385948, "clip_ratio/low_mean": 0.053072490729391575, "clip_ratio/low_min": 0.053072490729391575, "clip_ratio/high_mean": 0.07259316369891167, "clip_ratio/high_max": 0.07259316369891167, "clip_ratio/region_mean": 0.12566565442830324, "reward_total_mean": 0.7142574787139893, "reward_meter_mean": 0.8117817640304565, "reward_meter_std": 0.29429498314857483, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9583067893981934, "reward_repeat_soft_std": 0.023175112903118134, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.7142574787139893, "reward_total_composite_std": 0.12467237561941147} {"timestamp_utc": "2026-04-13T04:35:04Z", "mode": "train", "global_step": 2481, "epoch": 0.24922149673530888, "loss": 0.046, "grad_norm": 6.257554054260254, "learning_rate": 2.4848484848484848e-06, "num_tokens": 4597643.0, "completions/mean_length": 141.875, "completions/min_length": 134.0, "completions/max_length": 152.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 141.875, "completions/min_terminated_length": 134.0, "completions/max_terminated_length": 152.0, "rewards/meter/mean": 0.9149024486541748, "rewards/meter/std": 0.20511417090892792, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.95013827085495, "rewards/repeat_soft/std": 0.019110018387436867, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.7662199139595032, "rewards/total_composite/std": 0.10940930992364883, "reward": 0.7662199139595032, "reward_std": 0.10940933227539062, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10216344892978668, "sampling/sampling_logp_difference/max": 1.7398004531860352, "sampling/importance_sampling_ratio/min": 0.17555542290210724, "sampling/importance_sampling_ratio/mean": 1.0019971132278442, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5755359195172787, "clip_ratio/low_mean": 0.023473862558603287, "clip_ratio/low_min": 0.023473862558603287, "clip_ratio/high_mean": 0.08239458967000246, "clip_ratio/high_max": 0.08239458967000246, "clip_ratio/region_mean": 0.10586845222860575, "reward_total_mean": 0.7662199139595032, "reward_meter_mean": 0.9149024486541748, "reward_meter_std": 0.20511417090892792, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.95013827085495, "reward_repeat_soft_std": 0.019110018387436867, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.7662199139595032, "reward_total_composite_std": 0.10940930992364883} {"timestamp_utc": "2026-04-13T04:35:10Z", "mode": "train", "global_step": 2482, "epoch": 0.2493219487694626, "loss": 0.0337, "grad_norm": 16.276670455932617, "learning_rate": 2.481818181818182e-06, "num_tokens": 4599111.0, "completions/mean_length": 28.5, "completions/min_length": 24.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.5, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9772612452507019, "rewards/meter/std": 0.02432740293443203, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9526053071022034, "rewards/repeat_soft/std": 0.020112233236432076, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.0843462198972702, "rewards/total_composite/mean": 0.8005281090736389, "rewards/total_composite/std": 0.02510027587413788, "reward": 0.8005281090736389, "reward_std": 0.025100266560912132, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10656245052814484, "sampling/sampling_logp_difference/max": 1.139139175415039, "sampling/importance_sampling_ratio/min": 0.3200944662094116, "sampling/importance_sampling_ratio/mean": 1.0307708978652954, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8444065526127815, "clip_ratio/low_mean": 0.04592165909707546, "clip_ratio/low_min": 0.04592165909707546, "clip_ratio/high_mean": 0.03370160050690174, "clip_ratio/high_max": 0.03370160050690174, "clip_ratio/region_mean": 0.0796232596039772, "reward_total_mean": 0.8005281090736389, "reward_meter_mean": 0.9772612452507019, "reward_meter_std": 0.02432740293443203, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9526053071022034, "reward_repeat_soft_std": 0.020112233236432076, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.0843462198972702, "reward_total_composite_mean": 0.8005281090736389, "reward_total_composite_std": 0.02510027587413788} {"timestamp_utc": "2026-04-13T04:35:17Z", "mode": "train", "global_step": 2483, "epoch": 0.24942240080361627, "loss": 0.0433, "grad_norm": 11.60448169708252, "learning_rate": 2.478787878787879e-06, "num_tokens": 4600684.0, "completions/mean_length": 48.625, "completions/min_length": 43.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.625, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.32535481452941895, "rewards/meter/std": 0.34042245149612427, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9868667721748352, "rewards/repeat_soft/std": 0.01146503072232008, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.6335963010787964, "rewards/total_composite/std": 0.17630335688591003, "reward": 0.6335963010787964, "reward_std": 0.17630334198474884, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13459499180316925, "sampling/sampling_logp_difference/max": 2.472352981567383, "sampling/importance_sampling_ratio/min": 0.08438606560230255, "sampling/importance_sampling_ratio/mean": 0.9961189031600952, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5664241909980774, "clip_ratio/low_mean": 0.04732476198114455, "clip_ratio/low_min": 0.04732476198114455, "clip_ratio/high_mean": 0.0503608426079154, "clip_ratio/high_max": 0.0503608426079154, "clip_ratio/region_mean": 0.09768560458905995, "reward_total_mean": 0.6335963010787964, "reward_meter_mean": 0.32535481452941895, "reward_meter_std": 0.34042245149612427, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9868667721748352, "reward_repeat_soft_std": 0.01146503072232008, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.6335963010787964, "reward_total_composite_std": 0.17630335688591003} {"timestamp_utc": "2026-04-13T04:35:24Z", "mode": "train", "global_step": 2484, "epoch": 0.24952285283776995, "loss": 0.02, "grad_norm": 10.62725830078125, "learning_rate": 2.475757575757576e-06, "num_tokens": 4602857.0, "completions/mean_length": 61.625, "completions/min_length": 57.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.625, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9698739051818848, "rewards/meter/std": 0.03416714444756508, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9688839316368103, "rewards/repeat_soft/std": 0.025389473885297775, "rewards/judge_quality/mean": 0.5024999976158142, "rewards/judge_quality/std": 0.21224987506866455, "rewards/total_composite/mean": 0.8340816497802734, "rewards/total_composite/std": 0.06334654986858368, "reward": 0.8340816497802734, "reward_std": 0.06334655731916428, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12776648998260498, "sampling/sampling_logp_difference/max": 4.01802921295166, "sampling/importance_sampling_ratio/min": 0.017988381907343864, "sampling/importance_sampling_ratio/mean": 1.0179885625839233, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.737544059753418, "clip_ratio/low_mean": 0.08929902222007513, "clip_ratio/low_min": 0.08929902222007513, "clip_ratio/high_mean": 0.023443223908543587, "clip_ratio/high_max": 0.023443223908543587, "clip_ratio/region_mean": 0.11274224612861872, "reward_total_mean": 0.8340816497802734, "reward_meter_mean": 0.9698739051818848, "reward_meter_std": 0.03416714444756508, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9688839316368103, "reward_repeat_soft_std": 0.025389473885297775, "reward_judge_quality_mean": 0.5024999976158142, "reward_judge_quality_std": 0.21224987506866455, "reward_total_composite_mean": 0.8340816497802734, "reward_total_composite_std": 0.06334654986858368} {"timestamp_utc": "2026-04-13T04:35:30Z", "mode": "train", "global_step": 2485, "epoch": 0.24962330487192366, "loss": -0.002, "grad_norm": 18.464696884155273, "learning_rate": 2.472727272727273e-06, "num_tokens": 4604270.0, "completions/mean_length": 31.625, "completions/min_length": 28.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.625, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.48974335193634033, "rewards/meter/std": 0.42937734723091125, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.44624999165534973, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.600509524345398, "rewards/total_composite/std": 0.19172173738479614, "reward": 0.600509524345398, "reward_std": 0.19172175228595734, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13121305406093597, "sampling/sampling_logp_difference/max": 1.3578665256500244, "sampling/importance_sampling_ratio/min": 0.2572089433670044, "sampling/importance_sampling_ratio/mean": 1.009186863899231, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5580270029604435, "clip_ratio/low_mean": 0.04938556021079421, "clip_ratio/low_min": 0.04938556021079421, "clip_ratio/high_mean": 0.07162659103050828, "clip_ratio/high_max": 0.07162659103050828, "clip_ratio/region_mean": 0.12101215124130249, "reward_total_mean": 0.600509524345398, "reward_meter_mean": 0.48974335193634033, "reward_meter_std": 0.42937734723091125, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.44624999165534973, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.600509524345398, "reward_total_composite_std": 0.19172173738479614} {"timestamp_utc": "2026-04-13T04:35:42Z", "mode": "train", "global_step": 2486, "epoch": 0.24972375690607734, "loss": -0.1475, "grad_norm": 1.7810345888137817, "learning_rate": 2.46969696969697e-06, "num_tokens": 4605926.0, "completions/mean_length": 112.0, "completions/min_length": 49.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 54.857147216796875, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9744513034820557, "rewards/meter/std": 0.01926961913704872, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9664101004600525, "rewards/repeat_soft/std": 0.03526557236909866, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.13617216050624847, "rewards/total_composite/mean": 0.7121109962463379, "rewards/total_composite/std": 0.287882000207901, "reward": 0.7121109962463379, "reward_std": 0.2878819704055786, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10737340152263641, "sampling/sampling_logp_difference/max": 1.2248573303222656, "sampling/importance_sampling_ratio/min": 0.29379960894584656, "sampling/importance_sampling_ratio/mean": 1.0065069198608398, "sampling/importance_sampling_ratio/max": 1.8648046255111694, "entropy": 0.6149757876992226, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09222211176529527, "clip_ratio/high_max": 0.09222211176529527, "clip_ratio/region_mean": 0.09222211176529527, "reward_total_mean": 0.7121109962463379, "reward_meter_mean": 0.9744513034820557, "reward_meter_std": 0.01926961913704872, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9664101004600525, "reward_repeat_soft_std": 0.03526557236909866, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.13617216050624847, "reward_total_composite_mean": 0.7121109962463379, "reward_total_composite_std": 0.287882000207901} {"timestamp_utc": "2026-04-13T04:35:48Z", "mode": "train", "global_step": 2487, "epoch": 0.24982420894023105, "loss": -0.012, "grad_norm": 9.427655220031738, "learning_rate": 2.466666666666667e-06, "num_tokens": 4607685.0, "completions/mean_length": 59.875, "completions/min_length": 52.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.875, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9924087524414062, "rewards/meter/std": 0.0019691763445734978, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9377418160438538, "rewards/repeat_soft/std": 0.05014268681406975, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8197331428527832, "rewards/total_composite/std": 0.008252298459410667, "reward": 0.8197331428527832, "reward_std": 0.008252300322055817, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09854680299758911, "sampling/sampling_logp_difference/max": 1.4191546440124512, "sampling/importance_sampling_ratio/min": 0.2419184446334839, "sampling/importance_sampling_ratio/mean": 1.011020302772522, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5480860434472561, "clip_ratio/low_mean": 0.027910463977605104, "clip_ratio/low_min": 0.027910463977605104, "clip_ratio/high_mean": 0.038831149227917194, "clip_ratio/high_max": 0.038831149227917194, "clip_ratio/region_mean": 0.0667416132055223, "reward_total_mean": 0.8197331428527832, "reward_meter_mean": 0.9924087524414062, "reward_meter_std": 0.0019691763445734978, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9377418160438538, "reward_repeat_soft_std": 0.05014268681406975, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8197331428527832, "reward_total_composite_std": 0.008252298459410667} {"timestamp_utc": "2026-04-13T04:35:55Z", "mode": "train", "global_step": 2488, "epoch": 0.24992466097438473, "loss": 0.0205, "grad_norm": 7.571591377258301, "learning_rate": 2.463636363636364e-06, "num_tokens": 4609489.0, "completions/mean_length": 70.5, "completions/min_length": 63.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.5, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9932583570480347, "rewards/meter/std": 0.003525972366333008, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9898778200149536, "rewards/repeat_soft/std": 0.015135304071009159, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.8407040238380432, "rewards/total_composite/std": 0.05378998816013336, "reward": 0.8407040238380432, "reward_std": 0.05378998816013336, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13198819756507874, "sampling/sampling_logp_difference/max": 2.336684465408325, "sampling/importance_sampling_ratio/min": 0.09664754569530487, "sampling/importance_sampling_ratio/mean": 0.9928268194198608, "sampling/importance_sampling_ratio/max": 1.5855329036712646, "entropy": 0.955578438937664, "clip_ratio/low_mean": 0.0950588108971715, "clip_ratio/low_min": 0.0950588108971715, "clip_ratio/high_mean": 0.02083333395421505, "clip_ratio/high_max": 0.02083333395421505, "clip_ratio/region_mean": 0.11589214485138655, "reward_total_mean": 0.8407040238380432, "reward_meter_mean": 0.9932583570480347, "reward_meter_std": 0.003525972366333008, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9898778200149536, "reward_repeat_soft_std": 0.015135304071009159, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.8407040238380432, "reward_total_composite_std": 0.05378998816013336} {"timestamp_utc": "2026-04-13T04:36:03Z", "mode": "train", "global_step": 2489, "epoch": 0.2500251130085384, "loss": 0.0531, "grad_norm": 11.805115699768066, "learning_rate": 2.4606060606060607e-06, "num_tokens": 4611204.0, "completions/mean_length": 44.375, "completions/min_length": 40.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.375, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.963928759098053, "rewards/meter/std": 0.012718037702143192, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9337611198425293, "rewards/repeat_soft/std": 0.049221474677324295, "rewards/judge_quality/mean": 0.42124998569488525, "rewards/judge_quality/std": 0.06998724490404129, "rewards/total_composite/mean": 0.8035190105438232, "rewards/total_composite/std": 0.020111853256821632, "reward": 0.8035190105438232, "reward_std": 0.020111849531531334, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08806143701076508, "sampling/sampling_logp_difference/max": 1.4829649925231934, "sampling/importance_sampling_ratio/min": 0.22696374356746674, "sampling/importance_sampling_ratio/mean": 0.9957594871520996, "sampling/importance_sampling_ratio/max": 1.9378223419189453, "entropy": 0.33127764984965324, "clip_ratio/low_mean": 0.01918032392859459, "clip_ratio/low_min": 0.01918032392859459, "clip_ratio/high_mean": 0.06266164337284863, "clip_ratio/high_max": 0.06266164337284863, "clip_ratio/region_mean": 0.08184196730144322, "reward_total_mean": 0.8035190105438232, "reward_meter_mean": 0.963928759098053, "reward_meter_std": 0.012718037702143192, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9337611198425293, "reward_repeat_soft_std": 0.049221474677324295, "reward_judge_quality_mean": 0.42124998569488525, "reward_judge_quality_std": 0.06998724490404129, "reward_total_composite_mean": 0.8035190105438232, "reward_total_composite_std": 0.020111853256821632} {"timestamp_utc": "2026-04-13T04:36:12Z", "mode": "train", "global_step": 2490, "epoch": 0.2501255650426921, "loss": -0.0152, "grad_norm": 5.1337571144104, "learning_rate": 2.457575757575758e-06, "num_tokens": 4614339.0, "completions/mean_length": 181.875, "completions/min_length": 167.0, "completions/max_length": 194.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 181.875, "completions/min_terminated_length": 167.0, "completions/max_terminated_length": 194.0, "rewards/meter/mean": 0.9935956001281738, "rewards/meter/std": 0.004293540958315134, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8942214250564575, "rewards/repeat_soft/std": 0.036389682441949844, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.1249857097864151, "rewards/total_composite/mean": 0.7885401248931885, "rewards/total_composite/std": 0.04681865870952606, "reward": 0.7885401248931885, "reward_std": 0.04681865870952606, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07844807952642441, "sampling/sampling_logp_difference/max": 1.4548931121826172, "sampling/importance_sampling_ratio/min": 0.2334253191947937, "sampling/importance_sampling_ratio/mean": 1.0071775913238525, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4740164689719677, "clip_ratio/low_mean": 0.018111249431967735, "clip_ratio/low_min": 0.018111249431967735, "clip_ratio/high_mean": 0.05345220910385251, "clip_ratio/high_max": 0.05345220910385251, "clip_ratio/region_mean": 0.07156345853582025, "reward_total_mean": 0.7885401248931885, "reward_meter_mean": 0.9935956001281738, "reward_meter_std": 0.004293540958315134, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8942214250564575, "reward_repeat_soft_std": 0.036389682441949844, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.1249857097864151, "reward_total_composite_mean": 0.7885401248931885, "reward_total_composite_std": 0.04681865870952606} {"timestamp_utc": "2026-04-13T04:36:20Z", "mode": "train", "global_step": 2491, "epoch": 0.25022601707684583, "loss": 0.0208, "grad_norm": 12.764354705810547, "learning_rate": 2.454545454545455e-06, "num_tokens": 4616296.0, "completions/mean_length": 74.625, "completions/min_length": 67.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.625, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.8658237457275391, "rewards/meter/std": 0.23809374868869781, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9323520064353943, "rewards/repeat_soft/std": 0.03955959901213646, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.7654184103012085, "rewards/total_composite/std": 0.11534905433654785, "reward": 0.7654184103012085, "reward_std": 0.11534905433654785, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09871149063110352, "sampling/sampling_logp_difference/max": 1.8784160614013672, "sampling/importance_sampling_ratio/min": 0.1528320014476776, "sampling/importance_sampling_ratio/mean": 1.0037270784378052, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43410420045256615, "clip_ratio/low_mean": 0.024452269077301025, "clip_ratio/low_min": 0.024452269077301025, "clip_ratio/high_mean": 0.06371426326222718, "clip_ratio/high_max": 0.06371426326222718, "clip_ratio/region_mean": 0.0881665323395282, "reward_total_mean": 0.7654184103012085, "reward_meter_mean": 0.8658237457275391, "reward_meter_std": 0.23809374868869781, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9323520064353943, "reward_repeat_soft_std": 0.03955959901213646, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.7654184103012085, "reward_total_composite_std": 0.11534905433654785} {"timestamp_utc": "2026-04-13T04:36:32Z", "mode": "train", "global_step": 2492, "epoch": 0.2503264691109995, "loss": -0.1555, "grad_norm": 1.5160601139068604, "learning_rate": 2.4515151515151516e-06, "num_tokens": 4618071.0, "completions/mean_length": 114.875, "completions/min_length": 39.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 58.142860412597656, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9402479529380798, "rewards/meter/std": 0.14573991298675537, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9696272015571594, "rewards/repeat_soft/std": 0.026345033198595047, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.13617216050624847, "rewards/total_composite/mean": 0.719965398311615, "rewards/total_composite/std": 0.29098010063171387, "reward": 0.719965398311615, "reward_std": 0.2909800708293915, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10899663716554642, "sampling/sampling_logp_difference/max": 1.2413649559020996, "sampling/importance_sampling_ratio/min": 0.28898951411247253, "sampling/importance_sampling_ratio/mean": 1.0033940076828003, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5684530809521675, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.1112738698720932, "clip_ratio/high_max": 0.1112738698720932, "clip_ratio/region_mean": 0.1112738698720932, "reward_total_mean": 0.719965398311615, "reward_meter_mean": 0.9402479529380798, "reward_meter_std": 0.14573991298675537, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9696272015571594, "reward_repeat_soft_std": 0.026345033198595047, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.13617216050624847, "reward_total_composite_mean": 0.719965398311615, "reward_total_composite_std": 0.29098010063171387} {"timestamp_utc": "2026-04-13T04:36:38Z", "mode": "train", "global_step": 2493, "epoch": 0.2504269211451532, "loss": -0.0113, "grad_norm": 14.308506965637207, "learning_rate": 2.4484848484848485e-06, "num_tokens": 4619608.0, "completions/mean_length": 33.125, "completions/min_length": 30.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.125, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9692808389663696, "rewards/meter/std": 0.033472977578639984, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8095513582229614, "rewards/total_composite/std": 0.016014182940125465, "reward": 0.8095513582229614, "reward_std": 0.016014184802770615, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13957664370536804, "sampling/sampling_logp_difference/max": 1.4464070796966553, "sampling/importance_sampling_ratio/min": 0.23541460931301117, "sampling/importance_sampling_ratio/mean": 0.9864391684532166, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6799760945141315, "clip_ratio/low_mean": 0.050757577642798424, "clip_ratio/low_min": 0.050757577642798424, "clip_ratio/high_mean": 0.08505777455866337, "clip_ratio/high_max": 0.08505777455866337, "clip_ratio/region_mean": 0.1358153522014618, "reward_total_mean": 0.8095513582229614, "reward_meter_mean": 0.9692808389663696, "reward_meter_std": 0.033472977578639984, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8095513582229614, "reward_total_composite_std": 0.016014182940125465} {"timestamp_utc": "2026-04-13T04:36:52Z", "mode": "train", "global_step": 2494, "epoch": 0.2505273731793069, "loss": -0.1338, "grad_norm": 3.5990638732910156, "learning_rate": 2.4454545454545457e-06, "num_tokens": 4621354.0, "completions/mean_length": 123.25, "completions/min_length": 62.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 67.71428680419922, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.6415096521377563, "rewards/meter/std": 0.3772476613521576, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9634486436843872, "rewards/repeat_soft/std": 0.029060928151011467, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.5858038663864136, "rewards/total_composite/std": 0.2902892231941223, "reward": 0.5858038663864136, "reward_std": 0.2902892231941223, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11326488107442856, "sampling/sampling_logp_difference/max": 1.538179874420166, "sampling/importance_sampling_ratio/min": 0.21477167308330536, "sampling/importance_sampling_ratio/mean": 1.0260308980941772, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.444599874317646, "clip_ratio/low_mean": 0.03602181561291218, "clip_ratio/low_min": 0.03602181561291218, "clip_ratio/high_mean": 0.08110432978719473, "clip_ratio/high_max": 0.08110432978719473, "clip_ratio/region_mean": 0.1171261454001069, "reward_total_mean": 0.5858038663864136, "reward_meter_mean": 0.6415096521377563, "reward_meter_std": 0.3772476613521576, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9634486436843872, "reward_repeat_soft_std": 0.029060928151011467, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.5858038663864136, "reward_total_composite_std": 0.2902892231941223} {"timestamp_utc": "2026-04-13T04:37:03Z", "mode": "train", "global_step": 2495, "epoch": 0.25062782521346055, "loss": -0.1391, "grad_norm": 1.8679397106170654, "learning_rate": 2.4424242424242426e-06, "num_tokens": 4622906.0, "completions/mean_length": 107.0, "completions/min_length": 43.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 49.142860412597656, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.8660327196121216, "rewards/meter/std": 0.31120070815086365, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.99244225025177, "rewards/repeat_soft/std": 0.010293019004166126, "rewards/judge_quality/mean": 0.3712499737739563, "rewards/judge_quality/std": 0.1470119059085846, "rewards/total_composite/mean": 0.7117616534233093, "rewards/total_composite/std": 0.28858572244644165, "reward": 0.7117616534233093, "reward_std": 0.28858572244644165, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11021123826503754, "sampling/sampling_logp_difference/max": 1.3653926849365234, "sampling/importance_sampling_ratio/min": 0.25528040528297424, "sampling/importance_sampling_ratio/mean": 1.0195716619491577, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6929564997553825, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.07875306345522404, "clip_ratio/high_max": 0.07875306345522404, "clip_ratio/region_mean": 0.07875306345522404, "reward_total_mean": 0.7117616534233093, "reward_meter_mean": 0.8660327196121216, "reward_meter_std": 0.31120070815086365, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.99244225025177, "reward_repeat_soft_std": 0.010293019004166126, "reward_judge_quality_mean": 0.3712499737739563, "reward_judge_quality_std": 0.1470119059085846, "reward_total_composite_mean": 0.7117616534233093, "reward_total_composite_std": 0.28858572244644165} {"timestamp_utc": "2026-04-13T04:37:11Z", "mode": "train", "global_step": 2496, "epoch": 0.25072827724761426, "loss": 0.0467, "grad_norm": 6.14697790145874, "learning_rate": 2.4393939393939394e-06, "num_tokens": 4625339.0, "completions/mean_length": 115.125, "completions/min_length": 94.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.125, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.9866873025894165, "rewards/meter/std": 0.007359631359577179, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8320173025131226, "rewards/repeat_soft/std": 0.06165563687682152, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.313357412815094, "rewards/total_composite/mean": 0.8804609775543213, "rewards/total_composite/std": 0.09535226225852966, "reward": 0.8804609775543213, "reward_std": 0.09535227715969086, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08397436141967773, "sampling/sampling_logp_difference/max": 1.9584629535675049, "sampling/importance_sampling_ratio/min": 0.14107508957386017, "sampling/importance_sampling_ratio/mean": 1.010831594467163, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47661227360367775, "clip_ratio/low_mean": 0.030610764399170876, "clip_ratio/low_min": 0.030610764399170876, "clip_ratio/high_mean": 0.04818445770069957, "clip_ratio/high_max": 0.04818445770069957, "clip_ratio/region_mean": 0.07879522209987044, "reward_total_mean": 0.8804609775543213, "reward_meter_mean": 0.9866873025894165, "reward_meter_std": 0.007359631359577179, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8320173025131226, "reward_repeat_soft_std": 0.06165563687682152, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.313357412815094, "reward_total_composite_mean": 0.8804609775543213, "reward_total_composite_std": 0.09535226225852966} {"timestamp_utc": "2026-04-13T04:37:21Z", "mode": "train", "global_step": 2497, "epoch": 0.25082872928176797, "loss": 0.0213, "grad_norm": 4.2045159339904785, "learning_rate": 2.4363636363636366e-06, "num_tokens": 4628756.0, "completions/mean_length": 216.125, "completions/min_length": 195.0, "completions/max_length": 241.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 216.125, "completions/min_terminated_length": 195.0, "completions/max_terminated_length": 241.0, "rewards/meter/mean": 0.9930216073989868, "rewards/meter/std": 0.0024001665879040956, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9377031326293945, "rewards/repeat_soft/std": 0.024098435416817665, "rewards/judge_quality/mean": 0.20875000953674316, "rewards/judge_quality/std": 0.09657527506351471, "rewards/total_composite/mean": 0.7501300573348999, "rewards/total_composite/std": 0.029930096119642258, "reward": 0.7501300573348999, "reward_std": 0.029930099844932556, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1028878390789032, "sampling/sampling_logp_difference/max": 1.8587775230407715, "sampling/importance_sampling_ratio/min": 0.15586304664611816, "sampling/importance_sampling_ratio/mean": 1.0074071884155273, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6945557221770287, "clip_ratio/low_mean": 0.06656058086082339, "clip_ratio/low_min": 0.06656058086082339, "clip_ratio/high_mean": 0.027511416003108025, "clip_ratio/high_max": 0.027511416003108025, "clip_ratio/region_mean": 0.09407199686393142, "reward_total_mean": 0.7501300573348999, "reward_meter_mean": 0.9930216073989868, "reward_meter_std": 0.0024001665879040956, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9377031326293945, "reward_repeat_soft_std": 0.024098435416817665, "reward_judge_quality_mean": 0.20875000953674316, "reward_judge_quality_std": 0.09657527506351471, "reward_total_composite_mean": 0.7501300573348999, "reward_total_composite_std": 0.029930096119642258} {"timestamp_utc": "2026-04-13T04:37:33Z", "mode": "train", "global_step": 2498, "epoch": 0.2509291813159216, "loss": -0.2188, "grad_norm": 1.697572112083435, "learning_rate": 2.4333333333333335e-06, "num_tokens": 4631088.0, "completions/mean_length": 276.5, "completions/min_length": 123.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 135.1999969482422, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 147.0, "rewards/meter/mean": 0.7689954042434692, "rewards/meter/std": 0.41147762537002563, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.3204349875450134, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9655923843383789, "rewards/repeat_soft/std": 0.030582580715417862, "rewards/judge_quality/mean": 0.28125, "rewards/judge_quality/std": 0.1914931833744049, "rewards/total_composite/mean": 0.501258373260498, "rewards/total_composite/std": 0.4154341220855713, "reward": 0.501258373260498, "reward_std": 0.4154341220855713, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10731357336044312, "sampling/sampling_logp_difference/max": 4.643989086151123, "sampling/importance_sampling_ratio/min": 0.009619249030947685, "sampling/importance_sampling_ratio/mean": 1.0128470659255981, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44665153697133064, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.061957939993590117, "clip_ratio/high_max": 0.061957939993590117, "clip_ratio/region_mean": 0.061957939993590117, "reward_total_mean": 0.501258373260498, "reward_meter_mean": 0.7689954042434692, "reward_meter_std": 0.41147762537002563, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.3204349875450134, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9655923843383789, "reward_repeat_soft_std": 0.030582580715417862, "reward_judge_quality_mean": 0.28125, "reward_judge_quality_std": 0.1914931833744049, "reward_total_composite_mean": 0.501258373260498, "reward_total_composite_std": 0.4154341220855713} {"timestamp_utc": "2026-04-13T04:37:46Z", "mode": "train", "global_step": 2499, "epoch": 0.25102963335007533, "loss": -0.2145, "grad_norm": 1.6222667694091797, "learning_rate": 2.4303030303030307e-06, "num_tokens": 4633304.0, "completions/mean_length": 177.0, "completions/min_length": 119.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 129.1428680419922, "completions/min_terminated_length": 119.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.9085306525230408, "rewards/meter/std": 0.20440374314785004, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9506279230117798, "rewards/repeat_soft/std": 0.021839866414666176, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.23277443647384644, "rewards/total_composite/mean": 0.7458335757255554, "rewards/total_composite/std": 0.30554988980293274, "reward": 0.7458335757255554, "reward_std": 0.30554988980293274, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09904845058917999, "sampling/sampling_logp_difference/max": 3.6162643432617188, "sampling/importance_sampling_ratio/min": 0.026882914826273918, "sampling/importance_sampling_ratio/mean": 1.0078643560409546, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47048256918787956, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08754161419346929, "clip_ratio/high_max": 0.08754161419346929, "clip_ratio/region_mean": 0.08754161419346929, "reward_total_mean": 0.7458335757255554, "reward_meter_mean": 0.9085306525230408, "reward_meter_std": 0.20440374314785004, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9506279230117798, "reward_repeat_soft_std": 0.021839866414666176, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.23277443647384644, "reward_total_composite_mean": 0.7458335757255554, "reward_total_composite_std": 0.30554988980293274} {"timestamp_utc": "2026-04-13T04:37:53Z", "mode": "train", "global_step": 2500, "epoch": 0.25113008538422904, "loss": 0.0341, "grad_norm": 8.797061920166016, "learning_rate": 2.4272727272727276e-06, "num_tokens": 4635373.0, "completions/mean_length": 80.625, "completions/min_length": 76.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.625, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.8580757975578308, "rewards/meter/std": 0.16087524592876434, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9250797629356384, "rewards/repeat_soft/std": 0.03368333354592323, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7482670545578003, "rewards/total_composite/std": 0.08729062974452972, "reward": 0.7482670545578003, "reward_std": 0.08729064464569092, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10280919820070267, "sampling/sampling_logp_difference/max": 1.453404426574707, "sampling/importance_sampling_ratio/min": 0.23377306759357452, "sampling/importance_sampling_ratio/mean": 1.0074373483657837, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6718494780361652, "clip_ratio/low_mean": 0.03422619216144085, "clip_ratio/low_min": 0.03422619216144085, "clip_ratio/high_mean": 0.0832026181742549, "clip_ratio/high_max": 0.0832026181742549, "clip_ratio/region_mean": 0.11742881033569574, "reward_total_mean": 0.7482670545578003, "reward_meter_mean": 0.8580757975578308, "reward_meter_std": 0.16087524592876434, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9250797629356384, "reward_repeat_soft_std": 0.03368333354592323, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7482670545578003, "reward_total_composite_std": 0.08729062974452972} {"timestamp_utc": "2026-04-13T04:38:59Z", "mode": "eval", "global_step": 2500, "epoch": 0.25113008538422904, "eval_loss": NaN, "eval_runtime": 65.7334, "eval_samples_per_second": 1.217, "eval_steps_per_second": 0.152, "eval_num_tokens": 4635373.0, "eval_completions/mean_length": 119.725, "eval_completions/min_length": 40.9, "eval_completions/max_length": 277.9, "eval_completions/clipped_ratio": 0.05, "eval_completions/mean_terminated_length": 98.45535736083984, "eval_completions/min_terminated_length": 40.9, "eval_completions/max_terminated_length": 171.6, "eval_rewards/meter/mean": 0.8990675330162048, "eval_rewards/meter/std": 0.16890428140759467, "eval_rewards/count_adherence/mean": 0.9704166710376739, "eval_rewards/count_adherence/std": 0.06545893922448158, "eval_rewards/hard_gate/mean": 0.9625, "eval_rewards/hard_gate/std": 0.10606601536273956, "eval_rewards/repeat_soft/mean": 0.9198890149593353, "eval_rewards/repeat_soft/std": 0.06904629692435264, "eval_rewards/judge_quality/mean": 0.3883749932050705, "eval_rewards/judge_quality/std": 0.13099676556885242, "eval_rewards/total_composite/mean": 0.7408108651638031, "eval_rewards/total_composite/std": 0.14234417052939535, "eval_reward": 0.7408108651638031, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.05029936730861664, "eval_sampling/sampling_logp_difference/max": 0.8608535766601563, "eval_sampling/importance_sampling_ratio/min": 0.42630103826522825, "eval_sampling/importance_sampling_ratio/mean": 1.013002598285675, "eval_sampling/importance_sampling_ratio/max": 1.408586287498474, "eval_entropy": 0.5500797182321548, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7408108651638031, "eval_reward_meter_mean": 0.8990675330162048, "eval_reward_meter_std": 0.16890428140759467, "eval_reward_count_adherence_mean": 0.9704166710376739, "eval_reward_count_adherence_std": 0.06545893922448158, "eval_reward_hard_gate_mean": 0.9625, "eval_reward_hard_gate_std": 0.10606601536273956, "eval_reward_repeat_soft_mean": 0.9198890149593353, "eval_reward_repeat_soft_std": 0.06904629692435264, "eval_reward_judge_quality_mean": 0.3883749932050705, "eval_reward_judge_quality_std": 0.13099676556885242, "eval_reward_total_composite_mean": 0.7408108651638031, "eval_reward_total_composite_std": 0.14234417052939535} {"timestamp_utc": "2026-04-13T04:39:08Z", "mode": "train", "global_step": 2501, "epoch": 0.25123053741838275, "loss": 0.0007, "grad_norm": 13.846991539001465, "learning_rate": 2.4242424242424244e-06, "num_tokens": 4636913.0, "completions/mean_length": 30.5, "completions/min_length": 27.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.5, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9874102473258972, "rewards/meter/std": 0.005460042972117662, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9446625113487244, "rewards/repeat_soft/std": 0.03444957360625267, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.8193008899688721, "rewards/total_composite/std": 0.007159873843193054, "reward": 0.8193008899688721, "reward_std": 0.007159884087741375, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09631045162677765, "sampling/sampling_logp_difference/max": 0.8454030156135559, "sampling/importance_sampling_ratio/min": 0.4293842911720276, "sampling/importance_sampling_ratio/mean": 1.0169787406921387, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5979663468897343, "clip_ratio/low_mean": 0.03605510760098696, "clip_ratio/low_min": 0.03605510760098696, "clip_ratio/high_mean": 0.05341422278434038, "clip_ratio/high_max": 0.05341422278434038, "clip_ratio/region_mean": 0.08946933038532734, "reward_total_mean": 0.8193008899688721, "reward_meter_mean": 0.9874102473258972, "reward_meter_std": 0.005460042972117662, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9446625113487244, "reward_repeat_soft_std": 0.03444957360625267, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.8193008899688721, "reward_total_composite_std": 0.007159873843193054} {"timestamp_utc": "2026-04-13T04:39:15Z", "mode": "train", "global_step": 2502, "epoch": 0.2513309894525364, "loss": 0.0535, "grad_norm": 8.431437492370605, "learning_rate": 2.4212121212121216e-06, "num_tokens": 4639095.0, "completions/mean_length": 96.75, "completions/min_length": 84.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.75, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.8741240501403809, "rewards/meter/std": 0.1921091228723526, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9019339680671692, "rewards/repeat_soft/std": 0.03102767840027809, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.14520922303199768, "rewards/total_composite/mean": 0.753361701965332, "rewards/total_composite/std": 0.11156468093395233, "reward": 0.753361701965332, "reward_std": 0.11156468093395233, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08768386393785477, "sampling/sampling_logp_difference/max": 1.3029992580413818, "sampling/importance_sampling_ratio/min": 0.2717156410217285, "sampling/importance_sampling_ratio/mean": 1.0103394985198975, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.46041683852672577, "clip_ratio/low_mean": 0.03180879075080156, "clip_ratio/low_min": 0.03180879075080156, "clip_ratio/high_mean": 0.05575011996552348, "clip_ratio/high_max": 0.05575011996552348, "clip_ratio/region_mean": 0.08755891071632504, "reward_total_mean": 0.753361701965332, "reward_meter_mean": 0.8741240501403809, "reward_meter_std": 0.1921091228723526, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9019339680671692, "reward_repeat_soft_std": 0.03102767840027809, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.14520922303199768, "reward_total_composite_mean": 0.753361701965332, "reward_total_composite_std": 0.11156468093395233} {"timestamp_utc": "2026-04-13T04:39:22Z", "mode": "train", "global_step": 2503, "epoch": 0.2514314414866901, "loss": -0.0297, "grad_norm": 5.0001020431518555, "learning_rate": 2.4181818181818185e-06, "num_tokens": 4641448.0, "completions/mean_length": 114.125, "completions/min_length": 103.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.125, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.9838989973068237, "rewards/meter/std": 0.009158221073448658, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7112494707107544, "rewards/repeat_soft/std": 0.08013965934515, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.7643795013427734, "rewards/total_composite/std": 0.0268891341984272, "reward": 0.7643795013427734, "reward_std": 0.026889139786362648, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05799536406993866, "sampling/sampling_logp_difference/max": 1.8633384704589844, "sampling/importance_sampling_ratio/min": 0.1551537960767746, "sampling/importance_sampling_ratio/mean": 1.0119948387145996, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31187434308230877, "clip_ratio/low_mean": 0.03733431058935821, "clip_ratio/low_min": 0.03733431058935821, "clip_ratio/high_mean": 0.02089512418024242, "clip_ratio/high_max": 0.02089512418024242, "clip_ratio/region_mean": 0.05822943476960063, "reward_total_mean": 0.7643795013427734, "reward_meter_mean": 0.9838989973068237, "reward_meter_std": 0.009158221073448658, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7112494707107544, "reward_repeat_soft_std": 0.08013965934515, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.7643795013427734, "reward_total_composite_std": 0.0268891341984272} {"timestamp_utc": "2026-04-13T04:39:31Z", "mode": "train", "global_step": 2504, "epoch": 0.2515318935208438, "loss": 0.0496, "grad_norm": 4.977976322174072, "learning_rate": 2.4151515151515153e-06, "num_tokens": 4644843.0, "completions/mean_length": 215.375, "completions/min_length": 190.0, "completions/max_length": 240.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 215.375, "completions/min_terminated_length": 190.0, "completions/max_terminated_length": 240.0, "rewards/meter/mean": 0.9790027737617493, "rewards/meter/std": 0.007349715568125248, "rewards/count_adherence/mean": 0.9166666269302368, "rewards/count_adherence/std": 0.0890870913863182, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.745124101638794, "rewards/repeat_soft/std": 0.07679589837789536, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7785636782646179, "rewards/total_composite/std": 0.012716098688542843, "reward": 0.7785636782646179, "reward_std": 0.012716102413833141, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07500586658716202, "sampling/sampling_logp_difference/max": 1.8230485916137695, "sampling/importance_sampling_ratio/min": 0.16153255105018616, "sampling/importance_sampling_ratio/mean": 1.005527377128601, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4099346771836281, "clip_ratio/low_mean": 0.037552764639258385, "clip_ratio/low_min": 0.037552764639258385, "clip_ratio/high_mean": 0.026944348588585854, "clip_ratio/high_max": 0.026944348588585854, "clip_ratio/region_mean": 0.06449711322784424, "reward_total_mean": 0.7785636782646179, "reward_meter_mean": 0.9790027737617493, "reward_meter_std": 0.007349715568125248, "reward_count_adherence_mean": 0.9166666269302368, "reward_count_adherence_std": 0.0890870913863182, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.745124101638794, "reward_repeat_soft_std": 0.07679589837789536, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7785636782646179, "reward_total_composite_std": 0.012716098688542843} {"timestamp_utc": "2026-04-13T04:39:38Z", "mode": "train", "global_step": 2505, "epoch": 0.25163234555499747, "loss": 0.0166, "grad_norm": 15.696003913879395, "learning_rate": 2.412121212121212e-06, "num_tokens": 4646451.0, "completions/mean_length": 45.0, "completions/min_length": 42.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.0, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.8086195588111877, "rewards/meter/std": 0.2590908706188202, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9429525136947632, "rewards/repeat_soft/std": 0.08316176384687424, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404937028885, "rewards/total_composite/mean": 0.775049090385437, "rewards/total_composite/std": 0.15251317620277405, "reward": 0.775049090385437, "reward_std": 0.15251317620277405, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11017884314060211, "sampling/sampling_logp_difference/max": 1.687778115272522, "sampling/importance_sampling_ratio/min": 0.1849299520254135, "sampling/importance_sampling_ratio/mean": 1.0071016550064087, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49528633803129196, "clip_ratio/low_mean": 0.03600902762264013, "clip_ratio/low_min": 0.03600902762264013, "clip_ratio/high_mean": 0.050770816626027226, "clip_ratio/high_max": 0.050770816626027226, "clip_ratio/region_mean": 0.08677984424866736, "reward_total_mean": 0.775049090385437, "reward_meter_mean": 0.8086195588111877, "reward_meter_std": 0.2590908706188202, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9429525136947632, "reward_repeat_soft_std": 0.08316176384687424, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404937028885, "reward_total_composite_mean": 0.775049090385437, "reward_total_composite_std": 0.15251317620277405} {"timestamp_utc": "2026-04-13T04:39:49Z", "mode": "train", "global_step": 2506, "epoch": 0.2517327975891512, "loss": -0.172, "grad_norm": 3.4313557147979736, "learning_rate": 2.4090909090909094e-06, "num_tokens": 4648729.0, "completions/mean_length": 195.75, "completions/min_length": 127.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 150.57144165039062, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 166.0, "rewards/meter/mean": 0.7221765518188477, "rewards/meter/std": 0.373527467250824, "rewards/count_adherence/mean": 0.9166666269302368, "rewards/count_adherence/std": 0.0890870913863182, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8388617038726807, "rewards/repeat_soft/std": 0.1148960143327713, "rewards/judge_quality/mean": 0.22374999523162842, "rewards/judge_quality/std": 0.1387636661529541, "rewards/total_composite/mean": 0.5734073519706726, "rewards/total_composite/std": 0.271359384059906, "reward": 0.5734073519706726, "reward_std": 0.2713593542575836, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10420390218496323, "sampling/sampling_logp_difference/max": 2.7024528980255127, "sampling/importance_sampling_ratio/min": 0.06704086810350418, "sampling/importance_sampling_ratio/mean": 0.9965072274208069, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4045460745692253, "clip_ratio/low_mean": 0.00978915672749281, "clip_ratio/low_min": 0.00978915672749281, "clip_ratio/high_mean": 0.08034797105938196, "clip_ratio/high_max": 0.08034797105938196, "clip_ratio/region_mean": 0.09013712778687477, "reward_total_mean": 0.5734073519706726, "reward_meter_mean": 0.7221765518188477, "reward_meter_std": 0.373527467250824, "reward_count_adherence_mean": 0.9166666269302368, "reward_count_adherence_std": 0.0890870913863182, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8388617038726807, "reward_repeat_soft_std": 0.1148960143327713, "reward_judge_quality_mean": 0.22374999523162842, "reward_judge_quality_std": 0.1387636661529541, "reward_total_composite_mean": 0.5734073519706726, "reward_total_composite_std": 0.271359384059906} {"timestamp_utc": "2026-04-13T04:40:00Z", "mode": "train", "global_step": 2507, "epoch": 0.2518332496233049, "loss": -0.091, "grad_norm": 3.4562487602233887, "learning_rate": 2.4060606060606062e-06, "num_tokens": 4650628.0, "completions/mean_length": 92.375, "completions/min_length": 30.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 32.42857360839844, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.5023371577262878, "rewards/meter/std": 0.43182262778282166, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9671874642372131, "rewards/repeat_soft/std": 0.013258260674774647, "rewards/judge_quality/mean": 0.6312500238418579, "rewards/judge_quality/std": 0.33417007327079773, "rewards/total_composite/mean": 0.6290204524993896, "rewards/total_composite/std": 0.3063061833381653, "reward": 0.6290204524993896, "reward_std": 0.3063061833381653, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13517387211322784, "sampling/sampling_logp_difference/max": 0.9942779541015625, "sampling/importance_sampling_ratio/min": 0.3699904978275299, "sampling/importance_sampling_ratio/mean": 1.013830304145813, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.950906366109848, "clip_ratio/low_mean": 0.02797378972172737, "clip_ratio/low_min": 0.02797378972172737, "clip_ratio/high_mean": 0.09640587959438562, "clip_ratio/high_max": 0.09640587959438562, "clip_ratio/region_mean": 0.124379669316113, "reward_total_mean": 0.6290204524993896, "reward_meter_mean": 0.5023371577262878, "reward_meter_std": 0.43182262778282166, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9671874642372131, "reward_repeat_soft_std": 0.013258260674774647, "reward_judge_quality_mean": 0.6312500238418579, "reward_judge_quality_std": 0.33417007327079773, "reward_total_composite_mean": 0.6290204524993896, "reward_total_composite_std": 0.3063061833381653} {"timestamp_utc": "2026-04-13T04:40:08Z", "mode": "train", "global_step": 2508, "epoch": 0.25193370165745854, "loss": 0.0154, "grad_norm": 5.5031328201293945, "learning_rate": 2.403030303030303e-06, "num_tokens": 4652950.0, "completions/mean_length": 123.25, "completions/min_length": 113.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 123.25, "completions/min_terminated_length": 113.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.9830489158630371, "rewards/meter/std": 0.02694268524646759, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6571668386459351, "rewards/repeat_soft/std": 0.11413492262363434, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.7585886716842651, "rewards/total_composite/std": 0.028835909441113472, "reward": 0.7585886716842651, "reward_std": 0.028835903853178024, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07124271243810654, "sampling/sampling_logp_difference/max": 2.170926809310913, "sampling/importance_sampling_ratio/min": 0.11407184600830078, "sampling/importance_sampling_ratio/mean": 1.0059715509414673, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.42033880203962326, "clip_ratio/low_mean": 0.022208987269550562, "clip_ratio/low_min": 0.022208987269550562, "clip_ratio/high_mean": 0.03483458049595356, "clip_ratio/high_max": 0.03483458049595356, "clip_ratio/region_mean": 0.05704356776550412, "reward_total_mean": 0.7585886716842651, "reward_meter_mean": 0.9830489158630371, "reward_meter_std": 0.02694268524646759, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6571668386459351, "reward_repeat_soft_std": 0.11413492262363434, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.7585886716842651, "reward_total_composite_std": 0.028835909441113472} {"timestamp_utc": "2026-04-13T04:40:14Z", "mode": "train", "global_step": 2509, "epoch": 0.25203415369161225, "loss": 0.0336, "grad_norm": 15.247512817382812, "learning_rate": 2.4000000000000003e-06, "num_tokens": 4654416.0, "completions/mean_length": 30.25, "completions/min_length": 28.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.25, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9562304019927979, "rewards/meter/std": 0.08788283914327621, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9602810144424438, "rewards/repeat_soft/std": 0.006276086904108524, "rewards/judge_quality/mean": 0.41749998927116394, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.8015817403793335, "rewards/total_composite/std": 0.04151470586657524, "reward": 0.8015817403793335, "reward_std": 0.041514694690704346, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11086564511060715, "sampling/sampling_logp_difference/max": 1.8750207424163818, "sampling/importance_sampling_ratio/min": 0.1533517837524414, "sampling/importance_sampling_ratio/mean": 1.0082226991653442, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6758082434535027, "clip_ratio/low_mean": 0.02464978490024805, "clip_ratio/low_min": 0.02464978490024805, "clip_ratio/high_mean": 0.10278285993263125, "clip_ratio/high_max": 0.10278285993263125, "clip_ratio/region_mean": 0.1274326448328793, "reward_total_mean": 0.8015817403793335, "reward_meter_mean": 0.9562304019927979, "reward_meter_std": 0.08788283914327621, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9602810144424438, "reward_repeat_soft_std": 0.006276086904108524, "reward_judge_quality_mean": 0.41749998927116394, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.8015817403793335, "reward_total_composite_std": 0.04151470586657524} {"timestamp_utc": "2026-04-13T04:40:20Z", "mode": "train", "global_step": 2510, "epoch": 0.25213460572576596, "loss": 0.0396, "grad_norm": 13.959776878356934, "learning_rate": 2.396969696969697e-06, "num_tokens": 4655862.0, "completions/mean_length": 30.75, "completions/min_length": 27.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.75, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.8987478017807007, "rewards/meter/std": 0.11140866577625275, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.8891865611076355, "rewards/total_composite/std": 0.09767802804708481, "reward": 0.8891865611076355, "reward_std": 0.09767802804708481, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10451871901750565, "sampling/sampling_logp_difference/max": 1.6887931823730469, "sampling/importance_sampling_ratio/min": 0.18474233150482178, "sampling/importance_sampling_ratio/mean": 1.0234100818634033, "sampling/importance_sampling_ratio/max": 1.992801308631897, "entropy": 0.5010343566536903, "clip_ratio/low_mean": 0.0386904776096344, "clip_ratio/low_min": 0.0386904776096344, "clip_ratio/high_mean": 0.06078669521957636, "clip_ratio/high_max": 0.06078669521957636, "clip_ratio/region_mean": 0.09947717282921076, "reward_total_mean": 0.8891865611076355, "reward_meter_mean": 0.8987478017807007, "reward_meter_std": 0.11140866577625275, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.8891865611076355, "reward_total_composite_std": 0.09767802804708481} {"timestamp_utc": "2026-04-13T04:40:28Z", "mode": "train", "global_step": 2511, "epoch": 0.2522350577599196, "loss": -0.0262, "grad_norm": 4.995717525482178, "learning_rate": 2.393939393939394e-06, "num_tokens": 4658590.0, "completions/mean_length": 158.0, "completions/min_length": 134.0, "completions/max_length": 169.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 158.0, "completions/min_terminated_length": 134.0, "completions/max_terminated_length": 169.0, "rewards/meter/mean": 0.9933052062988281, "rewards/meter/std": 0.0015339415986090899, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9192066788673401, "rewards/repeat_soft/std": 0.041110944002866745, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8111580610275269, "rewards/total_composite/std": 0.009563669562339783, "reward": 0.8111580610275269, "reward_std": 0.009563679806888103, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08185718208551407, "sampling/sampling_logp_difference/max": 2.906374454498291, "sampling/importance_sampling_ratio/min": 0.05467359721660614, "sampling/importance_sampling_ratio/mean": 1.0085794925689697, "sampling/importance_sampling_ratio/max": 1.9676603078842163, "entropy": 0.4525647722184658, "clip_ratio/low_mean": 0.01691799843683839, "clip_ratio/low_min": 0.01691799843683839, "clip_ratio/high_mean": 0.05870663793757558, "clip_ratio/high_max": 0.05870663793757558, "clip_ratio/region_mean": 0.07562463637441397, "reward_total_mean": 0.8111580610275269, "reward_meter_mean": 0.9933052062988281, "reward_meter_std": 0.0015339415986090899, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9192066788673401, "reward_repeat_soft_std": 0.041110944002866745, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8111580610275269, "reward_total_composite_std": 0.009563669562339783} {"timestamp_utc": "2026-04-13T04:40:40Z", "mode": "train", "global_step": 2512, "epoch": 0.2523355097940733, "loss": -0.0322, "grad_norm": 5.736190319061279, "learning_rate": 2.3909090909090912e-06, "num_tokens": 4659985.0, "completions/mean_length": 94.375, "completions/min_length": 32.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 34.71428680419922, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.7182663083076477, "rewards/meter/std": 0.4137243330478668, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9662696123123169, "rewards/repeat_soft/std": 0.013869198970496655, "rewards/judge_quality/mean": 0.3812500238418579, "rewards/judge_quality/std": 0.13452960550785065, "rewards/total_composite/mean": 0.6372218132019043, "rewards/total_composite/std": 0.3084418475627899, "reward": 0.6372218132019043, "reward_std": 0.3084418475627899, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10562720894813538, "sampling/sampling_logp_difference/max": 2.134582996368408, "sampling/importance_sampling_ratio/min": 0.11829391121864319, "sampling/importance_sampling_ratio/mean": 1.015899658203125, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5531810633838177, "clip_ratio/low_mean": 0.013888888992369175, "clip_ratio/low_min": 0.013888888992369175, "clip_ratio/high_mean": 0.05436945683322847, "clip_ratio/high_max": 0.05436945683322847, "clip_ratio/region_mean": 0.06825834582559764, "reward_total_mean": 0.6372218132019043, "reward_meter_mean": 0.7182663083076477, "reward_meter_std": 0.4137243330478668, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9662696123123169, "reward_repeat_soft_std": 0.013869198970496655, "reward_judge_quality_mean": 0.3812500238418579, "reward_judge_quality_std": 0.13452960550785065, "reward_total_composite_mean": 0.6372218132019043, "reward_total_composite_std": 0.3084418475627899} {"timestamp_utc": "2026-04-13T04:40:48Z", "mode": "train", "global_step": 2513, "epoch": 0.25243596182822703, "loss": 0.0352, "grad_norm": 6.376730442047119, "learning_rate": 2.387878787878788e-06, "num_tokens": 4662650.0, "completions/mean_length": 147.125, "completions/min_length": 131.0, "completions/max_length": 158.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 147.125, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 158.0, "rewards/meter/mean": 0.9935442805290222, "rewards/meter/std": 0.004160004667937756, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9569412469863892, "rewards/repeat_soft/std": 0.0216886755079031, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.8124140501022339, "rewards/total_composite/std": 0.017157675698399544, "reward": 0.8124140501022339, "reward_std": 0.017157699912786484, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09752669930458069, "sampling/sampling_logp_difference/max": 5.000685691833496, "sampling/importance_sampling_ratio/min": 0.006733328104019165, "sampling/importance_sampling_ratio/mean": 1.0095436573028564, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5932120494544506, "clip_ratio/low_mean": 0.015031646005809307, "clip_ratio/low_min": 0.015031646005809307, "clip_ratio/high_mean": 0.07216452434659004, "clip_ratio/high_max": 0.07216452434659004, "clip_ratio/region_mean": 0.08719617035239935, "reward_total_mean": 0.8124140501022339, "reward_meter_mean": 0.9935442805290222, "reward_meter_std": 0.004160004667937756, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9569412469863892, "reward_repeat_soft_std": 0.0216886755079031, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.8124140501022339, "reward_total_composite_std": 0.017157675698399544} {"timestamp_utc": "2026-04-13T04:40:54Z", "mode": "train", "global_step": 2514, "epoch": 0.25253641386238074, "loss": 0.0161, "grad_norm": 10.247515678405762, "learning_rate": 2.3848484848484853e-06, "num_tokens": 4664406.0, "completions/mean_length": 58.5, "completions/min_length": 51.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.5, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.809094250202179, "rewards/meter/std": 0.34760168194770813, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9849206209182739, "rewards/repeat_soft/std": 0.01615806296467781, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7385845184326172, "rewards/total_composite/std": 0.1570044457912445, "reward": 0.7385845184326172, "reward_std": 0.1570044457912445, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10976941138505936, "sampling/sampling_logp_difference/max": 1.9199810028076172, "sampling/importance_sampling_ratio/min": 0.14660973846912384, "sampling/importance_sampling_ratio/mean": 1.0049558877944946, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5529225468635559, "clip_ratio/low_mean": 0.029886625707149506, "clip_ratio/low_min": 0.029886625707149506, "clip_ratio/high_mean": 0.08266707696020603, "clip_ratio/high_max": 0.08266707696020603, "clip_ratio/region_mean": 0.11255370266735554, "reward_total_mean": 0.7385845184326172, "reward_meter_mean": 0.809094250202179, "reward_meter_std": 0.34760168194770813, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9849206209182739, "reward_repeat_soft_std": 0.01615806296467781, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7385845184326172, "reward_total_composite_std": 0.1570044457912445} {"timestamp_utc": "2026-04-13T04:41:00Z", "mode": "train", "global_step": 2515, "epoch": 0.2526368658965344, "loss": 0.0544, "grad_norm": 10.09887981414795, "learning_rate": 2.381818181818182e-06, "num_tokens": 4666067.0, "completions/mean_length": 44.625, "completions/min_length": 40.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.625, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9481072425842285, "rewards/meter/std": 0.032998986542224884, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9710664749145508, "rewards/repeat_soft/std": 0.021402018144726753, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8008798956871033, "rewards/total_composite/std": 0.011765163391828537, "reward": 0.8008798956871033, "reward_std": 0.011765163391828537, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09649581462144852, "sampling/sampling_logp_difference/max": 1.8505709171295166, "sampling/importance_sampling_ratio/min": 0.15714742243289948, "sampling/importance_sampling_ratio/mean": 1.0049731731414795, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5002430789172649, "clip_ratio/low_mean": 0.02926736348308623, "clip_ratio/low_min": 0.02926736348308623, "clip_ratio/high_mean": 0.06631434801965952, "clip_ratio/high_max": 0.06631434801965952, "clip_ratio/region_mean": 0.09558171150274575, "reward_total_mean": 0.8008798956871033, "reward_meter_mean": 0.9481072425842285, "reward_meter_std": 0.032998986542224884, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9710664749145508, "reward_repeat_soft_std": 0.021402018144726753, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8008798956871033, "reward_total_composite_std": 0.011765163391828537} {"timestamp_utc": "2026-04-13T04:41:09Z", "mode": "train", "global_step": 2516, "epoch": 0.2527373179306881, "loss": -0.0198, "grad_norm": 5.414291858673096, "learning_rate": 2.378787878787879e-06, "num_tokens": 4669330.0, "completions/mean_length": 211.875, "completions/min_length": 186.0, "completions/max_length": 230.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 211.875, "completions/min_terminated_length": 186.0, "completions/max_terminated_length": 230.0, "rewards/meter/mean": 0.9593614935874939, "rewards/meter/std": 0.09769848734140396, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9126513004302979, "rewards/repeat_soft/std": 0.07120232284069061, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.7762278318405151, "rewards/total_composite/std": 0.052452199161052704, "reward": 0.7762278318405151, "reward_std": 0.05245219171047211, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09013622254133224, "sampling/sampling_logp_difference/max": 1.5692296028137207, "sampling/importance_sampling_ratio/min": 0.20820552110671997, "sampling/importance_sampling_ratio/mean": 1.0070775747299194, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6688544154167175, "clip_ratio/low_mean": 0.03485139459371567, "clip_ratio/low_min": 0.03485139459371567, "clip_ratio/high_mean": 0.05193554749712348, "clip_ratio/high_max": 0.05193554749712348, "clip_ratio/region_mean": 0.08678694209083915, "reward_total_mean": 0.7762278318405151, "reward_meter_mean": 0.9593614935874939, "reward_meter_std": 0.09769848734140396, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9126513004302979, "reward_repeat_soft_std": 0.07120232284069061, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.7762278318405151, "reward_total_composite_std": 0.052452199161052704} {"timestamp_utc": "2026-04-13T04:41:17Z", "mode": "train", "global_step": 2517, "epoch": 0.2528377699648418, "loss": 0.028, "grad_norm": 8.87062931060791, "learning_rate": 2.375757575757576e-06, "num_tokens": 4671030.0, "completions/mean_length": 53.5, "completions/min_length": 50.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.5, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9920440912246704, "rewards/meter/std": 0.006215184926986694, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8212122917175293, "rewards/repeat_soft/std": 0.03634548559784889, "rewards/judge_quality/mean": 0.5887500047683716, "rewards/judge_quality/std": 0.17266297340393066, "rewards/total_composite/mean": 0.8551660776138306, "rewards/total_composite/std": 0.04763684421777725, "reward": 0.8551660776138306, "reward_std": 0.047636840492486954, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0886416807770729, "sampling/sampling_logp_difference/max": 0.9762287139892578, "sampling/importance_sampling_ratio/min": 0.37672919034957886, "sampling/importance_sampling_ratio/mean": 1.0107351541519165, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5261145271360874, "clip_ratio/low_mean": 0.04235848970711231, "clip_ratio/low_min": 0.04235848970711231, "clip_ratio/high_mean": 0.05143370945006609, "clip_ratio/high_max": 0.05143370945006609, "clip_ratio/region_mean": 0.0937921991571784, "reward_total_mean": 0.8551660776138306, "reward_meter_mean": 0.9920440912246704, "reward_meter_std": 0.006215184926986694, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8212122917175293, "reward_repeat_soft_std": 0.03634548559784889, "reward_judge_quality_mean": 0.5887500047683716, "reward_judge_quality_std": 0.17266297340393066, "reward_total_composite_mean": 0.8551660776138306, "reward_total_composite_std": 0.04763684421777725} {"timestamp_utc": "2026-04-13T04:41:25Z", "mode": "train", "global_step": 2518, "epoch": 0.25293822199899546, "loss": 0.0522, "grad_norm": 9.936352729797363, "learning_rate": 2.372727272727273e-06, "num_tokens": 4672804.0, "completions/mean_length": 58.75, "completions/min_length": 53.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.75, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.794956386089325, "rewards/meter/std": 0.29142042994499207, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9676350355148315, "rewards/repeat_soft/std": 0.03329368680715561, "rewards/judge_quality/mean": 0.6349999904632568, "rewards/judge_quality/std": 0.21771542727947235, "rewards/total_composite/mean": 0.7949938774108887, "rewards/total_composite/std": 0.1591976284980774, "reward": 0.7949938774108887, "reward_std": 0.1591976284980774, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1148039698600769, "sampling/sampling_logp_difference/max": 1.8462109565734863, "sampling/importance_sampling_ratio/min": 0.15783408284187317, "sampling/importance_sampling_ratio/mean": 1.0141605138778687, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6506145671010017, "clip_ratio/low_mean": 0.042298982851207256, "clip_ratio/low_min": 0.042298982851207256, "clip_ratio/high_mean": 0.06320766778662801, "clip_ratio/high_max": 0.06320766778662801, "clip_ratio/region_mean": 0.10550665063783526, "reward_total_mean": 0.7949938774108887, "reward_meter_mean": 0.794956386089325, "reward_meter_std": 0.29142042994499207, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9676350355148315, "reward_repeat_soft_std": 0.03329368680715561, "reward_judge_quality_mean": 0.6349999904632568, "reward_judge_quality_std": 0.21771542727947235, "reward_total_composite_mean": 0.7949938774108887, "reward_total_composite_std": 0.1591976284980774} {"timestamp_utc": "2026-04-13T04:41:37Z", "mode": "train", "global_step": 2519, "epoch": 0.25303867403314917, "loss": -0.2312, "grad_norm": 1.446696400642395, "learning_rate": 2.36969696969697e-06, "num_tokens": 4675471.0, "completions/mean_length": 200.375, "completions/min_length": 149.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 155.85714721679688, "completions/min_terminated_length": 149.0, "completions/max_terminated_length": 168.0, "rewards/meter/mean": 0.9840430617332458, "rewards/meter/std": 0.01718652993440628, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.21380899846553802, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7319861054420471, "rewards/repeat_soft/std": 0.09887336194515228, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.7631429433822632, "rewards/total_composite/std": 0.066680409014225, "reward": 0.7631429433822632, "reward_std": 0.06668039411306381, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07471920549869537, "sampling/sampling_logp_difference/max": 2.1215803623199463, "sampling/importance_sampling_ratio/min": 0.11984208226203918, "sampling/importance_sampling_ratio/mean": 1.011339783668518, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3400323614478111, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06513560097664595, "clip_ratio/high_max": 0.06513560097664595, "clip_ratio/region_mean": 0.06513560097664595, "reward_total_mean": 0.7631429433822632, "reward_meter_mean": 0.9840430617332458, "reward_meter_std": 0.01718652993440628, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.21380899846553802, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7319861054420471, "reward_repeat_soft_std": 0.09887336194515228, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.7631429433822632, "reward_total_composite_std": 0.066680409014225} {"timestamp_utc": "2026-04-13T04:41:43Z", "mode": "train", "global_step": 2520, "epoch": 0.2531391260673029, "loss": 0.0383, "grad_norm": 13.529090881347656, "learning_rate": 2.3666666666666667e-06, "num_tokens": 4677077.0, "completions/mean_length": 54.75, "completions/min_length": 43.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.75, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.8703909516334534, "rewards/meter/std": 0.3228735327720642, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9899280667304993, "rewards/repeat_soft/std": 0.0072812410071492195, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.1865811049938202, "rewards/total_composite/mean": 0.8000437021255493, "rewards/total_composite/std": 0.1683090329170227, "reward": 0.8000437021255493, "reward_std": 0.1683090180158615, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11767502874135971, "sampling/sampling_logp_difference/max": 2.5586752891540527, "sampling/importance_sampling_ratio/min": 0.07740721106529236, "sampling/importance_sampling_ratio/mean": 0.9975564479827881, "sampling/importance_sampling_ratio/max": 1.846552848815918, "entropy": 0.6131271719932556, "clip_ratio/low_mean": 0.015350877307355404, "clip_ratio/low_min": 0.015350877307355404, "clip_ratio/high_mean": 0.11525585316121578, "clip_ratio/high_max": 0.11525585316121578, "clip_ratio/region_mean": 0.13060673046857119, "reward_total_mean": 0.8000437021255493, "reward_meter_mean": 0.8703909516334534, "reward_meter_std": 0.3228735327720642, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9899280667304993, "reward_repeat_soft_std": 0.0072812410071492195, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.1865811049938202, "reward_total_composite_mean": 0.8000437021255493, "reward_total_composite_std": 0.1683090329170227} {"timestamp_utc": "2026-04-13T04:41:52Z", "mode": "train", "global_step": 2521, "epoch": 0.25323957810145653, "loss": 0.065, "grad_norm": 4.377188205718994, "learning_rate": 2.363636363636364e-06, "num_tokens": 4680447.0, "completions/mean_length": 217.25, "completions/min_length": 196.0, "completions/max_length": 247.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 217.25, "completions/min_terminated_length": 196.0, "completions/max_terminated_length": 247.0, "rewards/meter/mean": 0.9868816137313843, "rewards/meter/std": 0.007720642257481813, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7697253823280334, "rewards/repeat_soft/std": 0.0838034525513649, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.1249857097864151, "rewards/total_composite/mean": 0.7705692648887634, "rewards/total_composite/std": 0.04074578732252121, "reward": 0.7705692648887634, "reward_std": 0.04074579104781151, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07549425214529037, "sampling/sampling_logp_difference/max": 1.9930723905563354, "sampling/importance_sampling_ratio/min": 0.13627609610557556, "sampling/importance_sampling_ratio/mean": 1.0115282535552979, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5063159614801407, "clip_ratio/low_mean": 0.02518952125683427, "clip_ratio/low_min": 0.02518952125683427, "clip_ratio/high_mean": 0.04969156254082918, "clip_ratio/high_max": 0.04969156254082918, "clip_ratio/region_mean": 0.07488108379766345, "reward_total_mean": 0.7705692648887634, "reward_meter_mean": 0.9868816137313843, "reward_meter_std": 0.007720642257481813, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7697253823280334, "reward_repeat_soft_std": 0.0838034525513649, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.1249857097864151, "reward_total_composite_mean": 0.7705692648887634, "reward_total_composite_std": 0.04074578732252121} {"timestamp_utc": "2026-04-13T04:41:59Z", "mode": "train", "global_step": 2522, "epoch": 0.25334003013561024, "loss": 0.0178, "grad_norm": 12.722992897033691, "learning_rate": 2.360606060606061e-06, "num_tokens": 4682080.0, "completions/mean_length": 59.125, "completions/min_length": 51.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.125, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.8692305088043213, "rewards/meter/std": 0.2140173763036728, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9948782920837402, "rewards/repeat_soft/std": 0.008467892184853554, "rewards/judge_quality/mean": 0.7437499761581421, "rewards/judge_quality/std": 0.2432481348514557, "rewards/total_composite/mean": 0.863766610622406, "rewards/total_composite/std": 0.0981164500117302, "reward": 0.863766610622406, "reward_std": 0.0981164500117302, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10847099125385284, "sampling/sampling_logp_difference/max": 1.8069299459457397, "sampling/importance_sampling_ratio/min": 0.16415733098983765, "sampling/importance_sampling_ratio/mean": 1.0178207159042358, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6704065352678299, "clip_ratio/low_mean": 0.04439657088369131, "clip_ratio/low_min": 0.04439657088369131, "clip_ratio/high_mean": 0.05245095491409302, "clip_ratio/high_max": 0.05245095491409302, "clip_ratio/region_mean": 0.09684752579778433, "reward_total_mean": 0.863766610622406, "reward_meter_mean": 0.8692305088043213, "reward_meter_std": 0.2140173763036728, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9948782920837402, "reward_repeat_soft_std": 0.008467892184853554, "reward_judge_quality_mean": 0.7437499761581421, "reward_judge_quality_std": 0.2432481348514557, "reward_total_composite_mean": 0.863766610622406, "reward_total_composite_std": 0.0981164500117302} {"timestamp_utc": "2026-04-13T04:42:10Z", "mode": "train", "global_step": 2523, "epoch": 0.25344048216976395, "loss": -0.188, "grad_norm": 2.172757148742676, "learning_rate": 2.3575757575757577e-06, "num_tokens": 4684127.0, "completions/mean_length": 146.875, "completions/min_length": 90.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 94.71428680419922, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.836024284362793, "rewards/meter/std": 0.18386003375053406, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9763479232788086, "rewards/repeat_soft/std": 0.029010990634560585, "rewards/judge_quality/mean": 0.38875001668930054, "rewards/judge_quality/std": 0.13767844438552856, "rewards/total_composite/mean": 0.6628347635269165, "rewards/total_composite/std": 0.2798914611339569, "reward": 0.6628347635269165, "reward_std": 0.2798914313316345, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11314321309328079, "sampling/sampling_logp_difference/max": 1.4547607898712158, "sampling/importance_sampling_ratio/min": 0.23345619440078735, "sampling/importance_sampling_ratio/mean": 1.0151495933532715, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5796908438205719, "clip_ratio/low_mean": 0.016129031777381897, "clip_ratio/low_min": 0.016129031777381897, "clip_ratio/high_mean": 0.09673627279698849, "clip_ratio/high_max": 0.09673627279698849, "clip_ratio/region_mean": 0.11286530457437038, "reward_total_mean": 0.6628347635269165, "reward_meter_mean": 0.836024284362793, "reward_meter_std": 0.18386003375053406, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9763479232788086, "reward_repeat_soft_std": 0.029010990634560585, "reward_judge_quality_mean": 0.38875001668930054, "reward_judge_quality_std": 0.13767844438552856, "reward_total_composite_mean": 0.6628347635269165, "reward_total_composite_std": 0.2798914611339569} {"timestamp_utc": "2026-04-13T04:42:17Z", "mode": "train", "global_step": 2524, "epoch": 0.25354093420391766, "loss": -0.0071, "grad_norm": 17.997716903686523, "learning_rate": 2.3545454545454545e-06, "num_tokens": 4685532.0, "completions/mean_length": 28.625, "completions/min_length": 24.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.625, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.5991314649581909, "rewards/meter/std": 0.3798092007637024, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.954291582107544, "rewards/repeat_soft/std": 0.023216886445879936, "rewards/judge_quality/mean": 0.6187499761581421, "rewards/judge_quality/std": 0.24976776540279388, "rewards/total_composite/mean": 0.7006633281707764, "rewards/total_composite/std": 0.21026621758937836, "reward": 0.7006633281707764, "reward_std": 0.21026621758937836, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12863954901695251, "sampling/sampling_logp_difference/max": 1.2505131959915161, "sampling/importance_sampling_ratio/min": 0.2863578200340271, "sampling/importance_sampling_ratio/mean": 1.0202022790908813, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6420756503939629, "clip_ratio/low_mean": 0.0416106628254056, "clip_ratio/low_min": 0.0416106628254056, "clip_ratio/high_mean": 0.04517932375892997, "clip_ratio/high_max": 0.04517932375892997, "clip_ratio/region_mean": 0.08678998658433557, "reward_total_mean": 0.7006633281707764, "reward_meter_mean": 0.5991314649581909, "reward_meter_std": 0.3798092007637024, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.954291582107544, "reward_repeat_soft_std": 0.023216886445879936, "reward_judge_quality_mean": 0.6187499761581421, "reward_judge_quality_std": 0.24976776540279388, "reward_total_composite_mean": 0.7006633281707764, "reward_total_composite_std": 0.21026621758937836} {"timestamp_utc": "2026-04-13T04:42:23Z", "mode": "train", "global_step": 2525, "epoch": 0.2536413862380713, "loss": 0.0207, "grad_norm": 8.952802658081055, "learning_rate": 2.3515151515151517e-06, "num_tokens": 4687330.0, "completions/mean_length": 59.75, "completions/min_length": 55.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.75, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9903208017349243, "rewards/meter/std": 0.007076128385961056, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9871302843093872, "rewards/repeat_soft/std": 0.013761091977357864, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.1538088023662567, "rewards/total_composite/mean": 0.8788573741912842, "rewards/total_composite/std": 0.04453744739294052, "reward": 0.8788573741912842, "reward_std": 0.044537436217069626, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08819496631622314, "sampling/sampling_logp_difference/max": 2.1920180320739746, "sampling/importance_sampling_ratio/min": 0.11169112473726273, "sampling/importance_sampling_ratio/mean": 1.0031620264053345, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4694015048444271, "clip_ratio/low_mean": 0.029272599145770073, "clip_ratio/low_min": 0.029272599145770073, "clip_ratio/high_mean": 0.0606142645701766, "clip_ratio/high_max": 0.0606142645701766, "clip_ratio/region_mean": 0.08988686371594667, "reward_total_mean": 0.8788573741912842, "reward_meter_mean": 0.9903208017349243, "reward_meter_std": 0.007076128385961056, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9871302843093872, "reward_repeat_soft_std": 0.013761091977357864, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.1538088023662567, "reward_total_composite_mean": 0.8788573741912842, "reward_total_composite_std": 0.04453744739294052} {"timestamp_utc": "2026-04-13T04:42:30Z", "mode": "train", "global_step": 2526, "epoch": 0.253741838272225, "loss": -0.0011, "grad_norm": 9.78816032409668, "learning_rate": 2.348484848484849e-06, "num_tokens": 4689180.0, "completions/mean_length": 80.25, "completions/min_length": 73.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.25, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.9713373184204102, "rewards/meter/std": 0.038865070790052414, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.987735390663147, "rewards/repeat_soft/std": 0.003829064778983593, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8118753433227539, "rewards/total_composite/std": 0.017545249313116074, "reward": 0.8118753433227539, "reward_std": 0.01754525862634182, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1279986947774887, "sampling/sampling_logp_difference/max": 1.7079700231552124, "sampling/importance_sampling_ratio/min": 0.18123331665992737, "sampling/importance_sampling_ratio/mean": 1.0014907121658325, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7194079011678696, "clip_ratio/low_mean": 0.02302631549537182, "clip_ratio/low_min": 0.02302631549537182, "clip_ratio/high_mean": 0.0987806273624301, "clip_ratio/high_max": 0.0987806273624301, "clip_ratio/region_mean": 0.12180694285780191, "reward_total_mean": 0.8118753433227539, "reward_meter_mean": 0.9713373184204102, "reward_meter_std": 0.038865070790052414, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.987735390663147, "reward_repeat_soft_std": 0.003829064778983593, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8118753433227539, "reward_total_composite_std": 0.017545249313116074} {"timestamp_utc": "2026-04-13T04:42:37Z", "mode": "train", "global_step": 2527, "epoch": 0.2538422903063787, "loss": -0.0215, "grad_norm": 8.541600227355957, "learning_rate": 2.345454545454546e-06, "num_tokens": 4691363.0, "completions/mean_length": 102.875, "completions/min_length": 89.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.875, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.8033545613288879, "rewards/meter/std": 0.21484704315662384, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8707033395767212, "rewards/repeat_soft/std": 0.10343915224075317, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.7247673869132996, "rewards/total_composite/std": 0.11094408482313156, "reward": 0.7247673869132996, "reward_std": 0.11094406247138977, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08350174129009247, "sampling/sampling_logp_difference/max": 1.8998656272888184, "sampling/importance_sampling_ratio/min": 0.1495887190103531, "sampling/importance_sampling_ratio/mean": 0.9977770447731018, "sampling/importance_sampling_ratio/max": 1.7054738998413086, "entropy": 0.45954959467053413, "clip_ratio/low_mean": 0.031835103407502174, "clip_ratio/low_min": 0.031835103407502174, "clip_ratio/high_mean": 0.052078885957598686, "clip_ratio/high_max": 0.052078885957598686, "clip_ratio/region_mean": 0.08391398936510086, "reward_total_mean": 0.7247673869132996, "reward_meter_mean": 0.8033545613288879, "reward_meter_std": 0.21484704315662384, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8707033395767212, "reward_repeat_soft_std": 0.10343915224075317, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.7247673869132996, "reward_total_composite_std": 0.11094408482313156} {"timestamp_utc": "2026-04-13T04:42:49Z", "mode": "train", "global_step": 2528, "epoch": 0.2539427423405324, "loss": -0.1189, "grad_norm": 2.5594425201416016, "learning_rate": 2.3424242424242427e-06, "num_tokens": 4692778.0, "completions/mean_length": 97.875, "completions/min_length": 31.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 38.71428680419922, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.36096248030662537, "rewards/meter/std": 0.35146477818489075, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9941118955612183, "rewards/repeat_soft/std": 0.00877163652330637, "rewards/judge_quality/mean": 0.5687500238418579, "rewards/judge_quality/std": 0.318856805562973, "rewards/total_composite/mean": 0.5363516211509705, "rewards/total_composite/std": 0.24883536994457245, "reward": 0.5363516211509705, "reward_std": 0.24883535504341125, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09749038517475128, "sampling/sampling_logp_difference/max": 1.3086719512939453, "sampling/importance_sampling_ratio/min": 0.2701786458492279, "sampling/importance_sampling_ratio/mean": 0.9986692667007446, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4023956134915352, "clip_ratio/low_mean": 0.01970046036876738, "clip_ratio/low_min": 0.01970046036876738, "clip_ratio/high_mean": 0.09403409250080585, "clip_ratio/high_max": 0.09403409250080585, "clip_ratio/region_mean": 0.11373455286957324, "reward_total_mean": 0.5363516211509705, "reward_meter_mean": 0.36096248030662537, "reward_meter_std": 0.35146477818489075, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9941118955612183, "reward_repeat_soft_std": 0.00877163652330637, "reward_judge_quality_mean": 0.5687500238418579, "reward_judge_quality_std": 0.318856805562973, "reward_total_composite_mean": 0.5363516211509705, "reward_total_composite_std": 0.24883536994457245} {"timestamp_utc": "2026-04-13T04:42:56Z", "mode": "train", "global_step": 2529, "epoch": 0.2540431943746861, "loss": 0.0078, "grad_norm": 6.632904529571533, "learning_rate": 2.3393939393939395e-06, "num_tokens": 4695217.0, "completions/mean_length": 127.875, "completions/min_length": 122.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.875, "completions/min_terminated_length": 122.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.6617375612258911, "rewards/meter/std": 0.35497984290122986, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9152214527130127, "rewards/repeat_soft/std": 0.05258796736598015, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.19949938356876373, "rewards/total_composite/mean": 0.6713041067123413, "rewards/total_composite/std": 0.16795380413532257, "reward": 0.6713041067123413, "reward_std": 0.16795380413532257, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09355498850345612, "sampling/sampling_logp_difference/max": 2.414447784423828, "sampling/importance_sampling_ratio/min": 0.08941670507192612, "sampling/importance_sampling_ratio/mean": 0.9932737946510315, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4999764710664749, "clip_ratio/low_mean": 0.027764528058469296, "clip_ratio/low_min": 0.027764528058469296, "clip_ratio/high_mean": 0.06951192952692509, "clip_ratio/high_max": 0.06951192952692509, "clip_ratio/region_mean": 0.09727645758539438, "reward_total_mean": 0.6713041067123413, "reward_meter_mean": 0.6617375612258911, "reward_meter_std": 0.35497984290122986, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9152214527130127, "reward_repeat_soft_std": 0.05258796736598015, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.19949938356876373, "reward_total_composite_mean": 0.6713041067123413, "reward_total_composite_std": 0.16795380413532257} {"timestamp_utc": "2026-04-13T04:43:07Z", "mode": "train", "global_step": 2530, "epoch": 0.2541436464088398, "loss": -0.2009, "grad_norm": 1.5880697965621948, "learning_rate": 2.3363636363636367e-06, "num_tokens": 4697342.0, "completions/mean_length": 152.625, "completions/min_length": 91.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 101.28572082519531, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.9015560746192932, "rewards/meter/std": 0.24889996647834778, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9546313285827637, "rewards/repeat_soft/std": 0.023666774854063988, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.7081834077835083, "rewards/total_composite/std": 0.2866520583629608, "reward": 0.7081834077835083, "reward_std": 0.2866520285606384, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10843461751937866, "sampling/sampling_logp_difference/max": 1.6334898471832275, "sampling/importance_sampling_ratio/min": 0.19524699449539185, "sampling/importance_sampling_ratio/mean": 0.9996922016143799, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6042540967464447, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.1121724545955658, "clip_ratio/high_max": 0.1121724545955658, "clip_ratio/region_mean": 0.1121724545955658, "reward_total_mean": 0.7081834077835083, "reward_meter_mean": 0.9015560746192932, "reward_meter_std": 0.24889996647834778, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9546313285827637, "reward_repeat_soft_std": 0.023666774854063988, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.7081834077835083, "reward_total_composite_std": 0.2866520583629608} {"timestamp_utc": "2026-04-13T04:43:14Z", "mode": "train", "global_step": 2531, "epoch": 0.25424409844299345, "loss": -0.0183, "grad_norm": 15.063182830810547, "learning_rate": 2.3333333333333336e-06, "num_tokens": 4698812.0, "completions/mean_length": 28.75, "completions/min_length": 24.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.75, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.8569042682647705, "rewards/meter/std": 0.3178962767124176, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9564294219017029, "rewards/repeat_soft/std": 0.017170162871479988, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.6875801086425781, "rewards/total_composite/std": 0.2793397605419159, "reward": 0.6875801086425781, "reward_std": 0.2793397605419159, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11081690341234207, "sampling/sampling_logp_difference/max": 1.3403091430664062, "sampling/importance_sampling_ratio/min": 0.2617647349834442, "sampling/importance_sampling_ratio/mean": 0.9972649812698364, "sampling/importance_sampling_ratio/max": 1.9451940059661865, "entropy": 0.5320217572152615, "clip_ratio/low_mean": 0.01923076994717121, "clip_ratio/low_min": 0.01923076994717121, "clip_ratio/high_mean": 0.09536862093955278, "clip_ratio/high_max": 0.09536862093955278, "clip_ratio/region_mean": 0.114599390886724, "reward_total_mean": 0.6875801086425781, "reward_meter_mean": 0.8569042682647705, "reward_meter_std": 0.3178962767124176, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9564294219017029, "reward_repeat_soft_std": 0.017170162871479988, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.6875801086425781, "reward_total_composite_std": 0.2793397605419159} {"timestamp_utc": "2026-04-13T04:43:21Z", "mode": "train", "global_step": 2532, "epoch": 0.25434455047714716, "loss": 0.0761, "grad_norm": 7.485983848571777, "learning_rate": 2.3303030303030304e-06, "num_tokens": 4701236.0, "completions/mean_length": 114.0, "completions/min_length": 104.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.0, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.9275772571563721, "rewards/meter/std": 0.15815040469169617, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9597738981246948, "rewards/repeat_soft/std": 0.02657943032681942, "rewards/judge_quality/mean": 0.35999998450279236, "rewards/judge_quality/std": 0.09165150672197342, "rewards/total_composite/mean": 0.7713871598243713, "rewards/total_composite/std": 0.06939758360385895, "reward": 0.7713871598243713, "reward_std": 0.06939759850502014, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11001569777727127, "sampling/sampling_logp_difference/max": 3.021847724914551, "sampling/importance_sampling_ratio/min": 0.048711132258176804, "sampling/importance_sampling_ratio/mean": 1.0065361261367798, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5742817595601082, "clip_ratio/low_mean": 0.04865248315036297, "clip_ratio/low_min": 0.04865248315036297, "clip_ratio/high_mean": 0.06534260883927345, "clip_ratio/high_max": 0.06534260883927345, "clip_ratio/region_mean": 0.11399509198963642, "reward_total_mean": 0.7713871598243713, "reward_meter_mean": 0.9275772571563721, "reward_meter_std": 0.15815040469169617, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9597738981246948, "reward_repeat_soft_std": 0.02657943032681942, "reward_judge_quality_mean": 0.35999998450279236, "reward_judge_quality_std": 0.09165150672197342, "reward_total_composite_mean": 0.7713871598243713, "reward_total_composite_std": 0.06939758360385895} {"timestamp_utc": "2026-04-13T04:43:33Z", "mode": "train", "global_step": 2533, "epoch": 0.25444500251130087, "loss": -0.2009, "grad_norm": 2.020829677581787, "learning_rate": 2.3272727272727277e-06, "num_tokens": 4703387.0, "completions/mean_length": 273.875, "completions/min_length": 123.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 131.0, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.4859682619571686, "rewards/meter/std": 0.4903384745121002, "rewards/count_adherence/mean": 0.5249999761581421, "rewards/count_adherence/std": 0.3845219910144806, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.860579252243042, "rewards/repeat_soft/std": 0.1286429911851883, "rewards/judge_quality/mean": 0.22624999284744263, "rewards/judge_quality/std": 0.17410485446453094, "rewards/total_composite/mean": 0.4044104814529419, "rewards/total_composite/std": 0.3551851212978363, "reward": 0.4044104814529419, "reward_std": 0.3551851212978363, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07913129031658173, "sampling/sampling_logp_difference/max": 2.715740919113159, "sampling/importance_sampling_ratio/min": 0.06615591794252396, "sampling/importance_sampling_ratio/mean": 1.0068963766098022, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2576449029147625, "clip_ratio/low_mean": 0.008130080997943878, "clip_ratio/low_min": 0.008130080997943878, "clip_ratio/high_mean": 0.032759728375822306, "clip_ratio/high_max": 0.032759728375822306, "clip_ratio/region_mean": 0.040889809373766184, "reward_total_mean": 0.4044104814529419, "reward_meter_mean": 0.4859682619571686, "reward_meter_std": 0.4903384745121002, "reward_count_adherence_mean": 0.5249999761581421, "reward_count_adherence_std": 0.3845219910144806, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.860579252243042, "reward_repeat_soft_std": 0.1286429911851883, "reward_judge_quality_mean": 0.22624999284744263, "reward_judge_quality_std": 0.17410485446453094, "reward_total_composite_mean": 0.4044104814529419, "reward_total_composite_std": 0.3551851212978363} {"timestamp_utc": "2026-04-13T04:43:41Z", "mode": "train", "global_step": 2534, "epoch": 0.2545454545454545, "loss": 0.0122, "grad_norm": 7.80953311920166, "learning_rate": 2.3242424242424245e-06, "num_tokens": 4705091.0, "completions/mean_length": 55.0, "completions/min_length": 53.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.0, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9731585383415222, "rewards/meter/std": 0.015292171388864517, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9026343822479248, "rewards/repeat_soft/std": 0.0617038793861866, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.8023098111152649, "rewards/total_composite/std": 0.018313780426979065, "reward": 0.8023098111152649, "reward_std": 0.01831376925110817, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07732319086790085, "sampling/sampling_logp_difference/max": 1.2684128284454346, "sampling/importance_sampling_ratio/min": 0.28127771615982056, "sampling/importance_sampling_ratio/mean": 1.004319190979004, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.403508473187685, "clip_ratio/low_mean": 0.027494027744978666, "clip_ratio/low_min": 0.027494027744978666, "clip_ratio/high_mean": 0.05732581065967679, "clip_ratio/high_max": 0.05732581065967679, "clip_ratio/region_mean": 0.08481983840465546, "reward_total_mean": 0.8023098111152649, "reward_meter_mean": 0.9731585383415222, "reward_meter_std": 0.015292171388864517, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9026343822479248, "reward_repeat_soft_std": 0.0617038793861866, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.8023098111152649, "reward_total_composite_std": 0.018313780426979065} {"timestamp_utc": "2026-04-13T04:43:53Z", "mode": "train", "global_step": 2535, "epoch": 0.2546459065796082, "loss": -0.0815, "grad_norm": 1.3516817092895508, "learning_rate": 2.3212121212121213e-06, "num_tokens": 4706391.0, "completions/mean_length": 83.5, "completions/min_length": 20.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 22.285715103149414, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.849799633026123, "rewards/meter/std": 0.33946678042411804, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9553493857383728, "rewards/repeat_soft/std": 0.01629802957177162, "rewards/judge_quality/mean": 0.3999999761581421, "rewards/judge_quality/std": 0.1414213478565216, "rewards/total_composite/mean": 0.714678168296814, "rewards/total_composite/std": 0.28887423872947693, "reward": 0.714678168296814, "reward_std": 0.28887423872947693, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09919595718383789, "sampling/sampling_logp_difference/max": 1.9619970321655273, "sampling/importance_sampling_ratio/min": 0.14057740569114685, "sampling/importance_sampling_ratio/mean": 1.0155240297317505, "sampling/importance_sampling_ratio/max": 1.9525916576385498, "entropy": 0.5148889720439911, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08388175396248698, "clip_ratio/high_max": 0.08388175396248698, "clip_ratio/region_mean": 0.08388175396248698, "reward_total_mean": 0.714678168296814, "reward_meter_mean": 0.849799633026123, "reward_meter_std": 0.33946678042411804, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9553493857383728, "reward_repeat_soft_std": 0.01629802957177162, "reward_judge_quality_mean": 0.3999999761581421, "reward_judge_quality_std": 0.1414213478565216, "reward_total_composite_mean": 0.714678168296814, "reward_total_composite_std": 0.28887423872947693} {"timestamp_utc": "2026-04-13T04:44:05Z", "mode": "train", "global_step": 2536, "epoch": 0.25474635861376194, "loss": -0.1145, "grad_norm": 1.3037101030349731, "learning_rate": 2.318181818181818e-06, "num_tokens": 4708027.0, "completions/mean_length": 165.5, "completions/min_length": 39.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 50.0, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.8870720267295837, "rewards/meter/std": 0.28512290120124817, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9275290966033936, "rewards/repeat_soft/std": 0.04577112942934036, "rewards/judge_quality/mean": 0.3425000011920929, "rewards/judge_quality/std": 0.18100906908512115, "rewards/total_composite/mean": 0.6582192182540894, "rewards/total_composite/std": 0.31428292393684387, "reward": 0.6582192182540894, "reward_std": 0.31428295373916626, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09695148468017578, "sampling/sampling_logp_difference/max": 1.8978281021118164, "sampling/importance_sampling_ratio/min": 0.14989382028579712, "sampling/importance_sampling_ratio/mean": 1.0039411783218384, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.424692515283823, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09292919281870127, "clip_ratio/high_max": 0.09292919281870127, "clip_ratio/region_mean": 0.09292919281870127, "reward_total_mean": 0.6582192182540894, "reward_meter_mean": 0.8870720267295837, "reward_meter_std": 0.28512290120124817, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9275290966033936, "reward_repeat_soft_std": 0.04577112942934036, "reward_judge_quality_mean": 0.3425000011920929, "reward_judge_quality_std": 0.18100906908512115, "reward_total_composite_mean": 0.6582192182540894, "reward_total_composite_std": 0.31428292393684387} {"timestamp_utc": "2026-04-13T04:44:11Z", "mode": "train", "global_step": 2537, "epoch": 0.25484681064791564, "loss": 0.0267, "grad_norm": 13.060196876525879, "learning_rate": 2.3151515151515154e-06, "num_tokens": 4709749.0, "completions/mean_length": 55.25, "completions/min_length": 52.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.25, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.945162296295166, "rewards/meter/std": 0.11926330626010895, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9782843589782715, "rewards/repeat_soft/std": 0.009558993391692638, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.8036514520645142, "rewards/total_composite/std": 0.0556802824139595, "reward": 0.8036514520645142, "reward_std": 0.0556802898645401, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10718151926994324, "sampling/sampling_logp_difference/max": 1.9467122554779053, "sampling/importance_sampling_ratio/min": 0.1427426040172577, "sampling/importance_sampling_ratio/mean": 1.0070997476577759, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5560395084321499, "clip_ratio/low_mean": 0.008928571827709675, "clip_ratio/low_min": 0.008928571827709675, "clip_ratio/high_mean": 0.09537779656238854, "clip_ratio/high_max": 0.09537779656238854, "clip_ratio/region_mean": 0.10430636839009821, "reward_total_mean": 0.8036514520645142, "reward_meter_mean": 0.945162296295166, "reward_meter_std": 0.11926330626010895, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9782843589782715, "reward_repeat_soft_std": 0.009558993391692638, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.8036514520645142, "reward_total_composite_std": 0.0556802824139595} {"timestamp_utc": "2026-04-13T04:44:18Z", "mode": "train", "global_step": 2538, "epoch": 0.2549472626820693, "loss": 0.006, "grad_norm": 10.701485633850098, "learning_rate": 2.3121212121212123e-06, "num_tokens": 4711552.0, "completions/mean_length": 52.375, "completions/min_length": 48.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.375, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9813612699508667, "rewards/meter/std": 0.01486609410494566, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9912011623382568, "rewards/repeat_soft/std": 0.010980244725942612, "rewards/judge_quality/mean": 0.768750011920929, "rewards/judge_quality/std": 0.21695540845394135, "rewards/total_composite/mean": 0.9213576316833496, "rewards/total_composite/std": 0.06996328383684158, "reward": 0.9213576316833496, "reward_std": 0.06996329873800278, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09852279722690582, "sampling/sampling_logp_difference/max": 1.8510586023330688, "sampling/importance_sampling_ratio/min": 0.1570708006620407, "sampling/importance_sampling_ratio/mean": 1.006247639656067, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4635300152003765, "clip_ratio/low_mean": 0.03585233259946108, "clip_ratio/low_min": 0.03585233259946108, "clip_ratio/high_mean": 0.07646745396777987, "clip_ratio/high_max": 0.07646745396777987, "clip_ratio/region_mean": 0.11231978656724095, "reward_total_mean": 0.9213576316833496, "reward_meter_mean": 0.9813612699508667, "reward_meter_std": 0.01486609410494566, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9912011623382568, "reward_repeat_soft_std": 0.010980244725942612, "reward_judge_quality_mean": 0.768750011920929, "reward_judge_quality_std": 0.21695540845394135, "reward_total_composite_mean": 0.9213576316833496, "reward_total_composite_std": 0.06996328383684158} {"timestamp_utc": "2026-04-13T04:44:25Z", "mode": "train", "global_step": 2539, "epoch": 0.255047714716223, "loss": 0.0747, "grad_norm": 7.338036060333252, "learning_rate": 2.309090909090909e-06, "num_tokens": 4713393.0, "completions/mean_length": 80.125, "completions/min_length": 66.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.125, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.9718493819236755, "rewards/meter/std": 0.029150376096367836, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9727238416671753, "rewards/repeat_soft/std": 0.017104173079133034, "rewards/judge_quality/mean": 0.5362499952316284, "rewards/judge_quality/std": 0.16070716083049774, "rewards/total_composite/mean": 0.8454796075820923, "rewards/total_composite/std": 0.05695231258869171, "reward": 0.8454796075820923, "reward_std": 0.056952331215143204, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0951160341501236, "sampling/sampling_logp_difference/max": 1.669797658920288, "sampling/importance_sampling_ratio/min": 0.18828515708446503, "sampling/importance_sampling_ratio/mean": 0.9961113929748535, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45713021233677864, "clip_ratio/low_mean": 0.051872171461582184, "clip_ratio/low_min": 0.051872171461582184, "clip_ratio/high_mean": 0.040494365617632866, "clip_ratio/high_max": 0.040494365617632866, "clip_ratio/region_mean": 0.09236653707921505, "reward_total_mean": 0.8454796075820923, "reward_meter_mean": 0.9718493819236755, "reward_meter_std": 0.029150376096367836, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9727238416671753, "reward_repeat_soft_std": 0.017104173079133034, "reward_judge_quality_mean": 0.5362499952316284, "reward_judge_quality_std": 0.16070716083049774, "reward_total_composite_mean": 0.8454796075820923, "reward_total_composite_std": 0.05695231258869171} {"timestamp_utc": "2026-04-13T04:44:37Z", "mode": "train", "global_step": 2540, "epoch": 0.2551481667503767, "loss": -0.1617, "grad_norm": 1.5544074773788452, "learning_rate": 2.306060606060606e-06, "num_tokens": 4715194.0, "completions/mean_length": 120.125, "completions/min_length": 58.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 64.14286041259766, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.8737964630126953, "rewards/meter/std": 0.339872807264328, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9656262993812561, "rewards/repeat_soft/std": 0.03932073339819908, "rewards/judge_quality/mean": 0.5687500238418579, "rewards/judge_quality/std": 0.2931327223777771, "rewards/total_composite/mean": 0.775902271270752, "rewards/total_composite/std": 0.319367915391922, "reward": 0.775902271270752, "reward_std": 0.3193678855895996, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09246061742305756, "sampling/sampling_logp_difference/max": 1.5321595668792725, "sampling/importance_sampling_ratio/min": 0.2160685658454895, "sampling/importance_sampling_ratio/mean": 1.007082462310791, "sampling/importance_sampling_ratio/max": 1.9417321681976318, "entropy": 0.4425166882574558, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09739688318222761, "clip_ratio/high_max": 0.09739688318222761, "clip_ratio/region_mean": 0.09739688318222761, "reward_total_mean": 0.775902271270752, "reward_meter_mean": 0.8737964630126953, "reward_meter_std": 0.339872807264328, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9656262993812561, "reward_repeat_soft_std": 0.03932073339819908, "reward_judge_quality_mean": 0.5687500238418579, "reward_judge_quality_std": 0.2931327223777771, "reward_total_composite_mean": 0.775902271270752, "reward_total_composite_std": 0.319367915391922} {"timestamp_utc": "2026-04-13T04:44:45Z", "mode": "train", "global_step": 2541, "epoch": 0.25524861878453037, "loss": 0.0487, "grad_norm": 14.813992500305176, "learning_rate": 2.303030303030303e-06, "num_tokens": 4716713.0, "completions/mean_length": 29.875, "completions/min_length": 26.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.875, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9742354154586792, "rewards/meter/std": 0.018750321120023727, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9585226774215698, "rewards/repeat_soft/std": 0.011249415576457977, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8136332035064697, "rewards/total_composite/std": 0.008942008018493652, "reward": 0.8136332035064697, "reward_std": 0.00894201546907425, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11110586673021317, "sampling/sampling_logp_difference/max": 1.3541841506958008, "sampling/importance_sampling_ratio/min": 0.2581578195095062, "sampling/importance_sampling_ratio/mean": 0.98951256275177, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5277997329831123, "clip_ratio/low_mean": 0.023958333767950535, "clip_ratio/low_min": 0.023958333767950535, "clip_ratio/high_mean": 0.07925938349217176, "clip_ratio/high_max": 0.07925938349217176, "clip_ratio/region_mean": 0.1032177172601223, "reward_total_mean": 0.8136332035064697, "reward_meter_mean": 0.9742354154586792, "reward_meter_std": 0.018750321120023727, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9585226774215698, "reward_repeat_soft_std": 0.011249415576457977, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8136332035064697, "reward_total_composite_std": 0.008942008018493652} {"timestamp_utc": "2026-04-13T04:44:53Z", "mode": "train", "global_step": 2542, "epoch": 0.2553490708186841, "loss": 0.0469, "grad_norm": 6.564174652099609, "learning_rate": 2.3000000000000004e-06, "num_tokens": 4719147.0, "completions/mean_length": 136.25, "completions/min_length": 131.0, "completions/max_length": 148.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.25, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 148.0, "rewards/meter/mean": 0.93086177110672, "rewards/meter/std": 0.11988068372011185, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7050715684890747, "rewards/repeat_soft/std": 0.1086687445640564, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.765394926071167, "rewards/total_composite/std": 0.055736787617206573, "reward": 0.765394926071167, "reward_std": 0.05573679879307747, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0682758018374443, "sampling/sampling_logp_difference/max": 1.6301732063293457, "sampling/importance_sampling_ratio/min": 0.19589562714099884, "sampling/importance_sampling_ratio/mean": 1.0092452764511108, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36000086180865765, "clip_ratio/low_mean": 0.015595261007547379, "clip_ratio/low_min": 0.015595261007547379, "clip_ratio/high_mean": 0.04129469965118915, "clip_ratio/high_max": 0.04129469965118915, "clip_ratio/region_mean": 0.05688996065873653, "reward_total_mean": 0.765394926071167, "reward_meter_mean": 0.93086177110672, "reward_meter_std": 0.11988068372011185, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7050715684890747, "reward_repeat_soft_std": 0.1086687445640564, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.765394926071167, "reward_total_composite_std": 0.055736787617206573} {"timestamp_utc": "2026-04-13T04:45:01Z", "mode": "train", "global_step": 2543, "epoch": 0.2554495228528378, "loss": 0.0193, "grad_norm": 8.83467960357666, "learning_rate": 2.2969696969696973e-06, "num_tokens": 4721631.0, "completions/mean_length": 115.5, "completions/min_length": 108.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.5, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.9082629680633545, "rewards/meter/std": 0.2445346862077713, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9721344709396362, "rewards/repeat_soft/std": 0.01539842039346695, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7755568027496338, "rewards/total_composite/std": 0.10909964889287949, "reward": 0.7755568027496338, "reward_std": 0.10909964889287949, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11899355798959732, "sampling/sampling_logp_difference/max": 1.8485193252563477, "sampling/importance_sampling_ratio/min": 0.15747015178203583, "sampling/importance_sampling_ratio/mean": 1.0156800746917725, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7251079231500626, "clip_ratio/low_mean": 0.03192452620714903, "clip_ratio/low_min": 0.03192452620714903, "clip_ratio/high_mean": 0.0745719843544066, "clip_ratio/high_max": 0.0745719843544066, "clip_ratio/region_mean": 0.10649651056155562, "reward_total_mean": 0.7755568027496338, "reward_meter_mean": 0.9082629680633545, "reward_meter_std": 0.2445346862077713, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9721344709396362, "reward_repeat_soft_std": 0.01539842039346695, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7755568027496338, "reward_total_composite_std": 0.10909964889287949} {"timestamp_utc": "2026-04-13T04:45:09Z", "mode": "train", "global_step": 2544, "epoch": 0.25554997488699144, "loss": -0.0158, "grad_norm": 6.748255252838135, "learning_rate": 2.293939393939394e-06, "num_tokens": 4724249.0, "completions/mean_length": 132.25, "completions/min_length": 110.0, "completions/max_length": 171.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.25, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 171.0, "rewards/meter/mean": 0.7317228317260742, "rewards/meter/std": 0.2798691391944885, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.952125608921051, "rewards/repeat_soft/std": 0.028242889791727066, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.6793003678321838, "rewards/total_composite/std": 0.1268559843301773, "reward": 0.6793003678321838, "reward_std": 0.1268559694290161, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1095680296421051, "sampling/sampling_logp_difference/max": 4.78592586517334, "sampling/importance_sampling_ratio/min": 0.008346391841769218, "sampling/importance_sampling_ratio/mean": 0.9996903538703918, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47663022577762604, "clip_ratio/low_mean": 0.023578811436891556, "clip_ratio/low_min": 0.023578811436891556, "clip_ratio/high_mean": 0.09030005987733603, "clip_ratio/high_max": 0.09030005987733603, "clip_ratio/region_mean": 0.11387887131422758, "reward_total_mean": 0.6793003678321838, "reward_meter_mean": 0.7317228317260742, "reward_meter_std": 0.2798691391944885, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.952125608921051, "reward_repeat_soft_std": 0.028242889791727066, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.6793003678321838, "reward_total_composite_std": 0.1268559843301773} {"timestamp_utc": "2026-04-13T04:45:17Z", "mode": "train", "global_step": 2545, "epoch": 0.25565042692114515, "loss": 0.0409, "grad_norm": 5.7433085441589355, "learning_rate": 2.2909090909090913e-06, "num_tokens": 4727151.0, "completions/mean_length": 160.75, "completions/min_length": 138.0, "completions/max_length": 178.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 160.75, "completions/min_terminated_length": 138.0, "completions/max_terminated_length": 178.0, "rewards/meter/mean": 0.9885917901992798, "rewards/meter/std": 0.011667967773973942, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8797789812088013, "rewards/repeat_soft/std": 0.05145818740129471, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7859691381454468, "rewards/total_composite/std": 0.02141091786324978, "reward": 0.7859691381454468, "reward_std": 0.02141091786324978, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08035379648208618, "sampling/sampling_logp_difference/max": 3.1256046295166016, "sampling/importance_sampling_ratio/min": 0.043910376727581024, "sampling/importance_sampling_ratio/mean": 1.0079447031021118, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44004473462700844, "clip_ratio/low_mean": 0.04071488883346319, "clip_ratio/low_min": 0.04071488883346319, "clip_ratio/high_mean": 0.034016046207398176, "clip_ratio/high_max": 0.034016046207398176, "clip_ratio/region_mean": 0.07473093504086137, "reward_total_mean": 0.7859691381454468, "reward_meter_mean": 0.9885917901992798, "reward_meter_std": 0.011667967773973942, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8797789812088013, "reward_repeat_soft_std": 0.05145818740129471, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7859691381454468, "reward_total_composite_std": 0.02141091786324978} {"timestamp_utc": "2026-04-13T04:45:28Z", "mode": "train", "global_step": 2546, "epoch": 0.25575087895529885, "loss": -0.0408, "grad_norm": 1.8030611276626587, "learning_rate": 2.287878787878788e-06, "num_tokens": 4728862.0, "completions/mean_length": 116.875, "completions/min_length": 40.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 60.42857360839844, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 161.0, "rewards/meter/mean": 0.8312493562698364, "rewards/meter/std": 0.3276200294494629, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.3720119297504425, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8039320707321167, "rewards/repeat_soft/std": 0.13450796902179718, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.13274572789669037, "rewards/total_composite/mean": 0.6895804405212402, "rewards/total_composite/std": 0.2125154435634613, "reward": 0.6895804405212402, "reward_std": 0.2125154435634613, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.058823924511671066, "sampling/sampling_logp_difference/max": 1.716418743133545, "sampling/importance_sampling_ratio/min": 0.1797085851430893, "sampling/importance_sampling_ratio/mean": 1.0191271305084229, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31653861701488495, "clip_ratio/low_mean": 0.00776397529989481, "clip_ratio/low_min": 0.00776397529989481, "clip_ratio/high_mean": 0.037198451114818454, "clip_ratio/high_max": 0.037198451114818454, "clip_ratio/region_mean": 0.044962426414713264, "reward_total_mean": 0.6895804405212402, "reward_meter_mean": 0.8312493562698364, "reward_meter_std": 0.3276200294494629, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.3720119297504425, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8039320707321167, "reward_repeat_soft_std": 0.13450796902179718, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.13274572789669037, "reward_total_composite_mean": 0.6895804405212402, "reward_total_composite_std": 0.2125154435634613} {"timestamp_utc": "2026-04-13T04:45:35Z", "mode": "train", "global_step": 2547, "epoch": 0.25585133098945256, "loss": 0.0313, "grad_norm": 10.964510917663574, "learning_rate": 2.284848484848485e-06, "num_tokens": 4730567.0, "completions/mean_length": 53.125, "completions/min_length": 46.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.125, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9886715412139893, "rewards/meter/std": 0.006994474213570356, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9666673541069031, "rewards/repeat_soft/std": 0.024886026978492737, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.9300689101219177, "rewards/total_composite/std": 0.069536492228508, "reward": 0.9300689101219177, "reward_std": 0.06953649967908859, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10462222248315811, "sampling/sampling_logp_difference/max": 1.340104579925537, "sampling/importance_sampling_ratio/min": 0.2618182897567749, "sampling/importance_sampling_ratio/mean": 1.014663577079773, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5387305803596973, "clip_ratio/low_mean": 0.015877695754170418, "clip_ratio/low_min": 0.015877695754170418, "clip_ratio/high_mean": 0.08193999342620373, "clip_ratio/high_max": 0.08193999342620373, "clip_ratio/region_mean": 0.09781768918037415, "reward_total_mean": 0.9300689101219177, "reward_meter_mean": 0.9886715412139893, "reward_meter_std": 0.006994474213570356, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9666673541069031, "reward_repeat_soft_std": 0.024886026978492737, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.9300689101219177, "reward_total_composite_std": 0.069536492228508} {"timestamp_utc": "2026-04-13T04:45:42Z", "mode": "train", "global_step": 2548, "epoch": 0.2559517830236062, "loss": 0.0201, "grad_norm": 6.133282661437988, "learning_rate": 2.281818181818182e-06, "num_tokens": 4733044.0, "completions/mean_length": 125.625, "completions/min_length": 118.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.625, "completions/min_terminated_length": 118.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.9734506607055664, "rewards/meter/std": 0.019772527739405632, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8855016231536865, "rewards/repeat_soft/std": 0.06458554416894913, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.20860078930854797, "rewards/total_composite/mean": 0.8086029291152954, "rewards/total_composite/std": 0.06107509136199951, "reward": 0.8086029291152954, "reward_std": 0.06107509881258011, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08445681631565094, "sampling/sampling_logp_difference/max": 2.3314359188079834, "sampling/importance_sampling_ratio/min": 0.09715613722801208, "sampling/importance_sampling_ratio/mean": 0.9992751479148865, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3927628807723522, "clip_ratio/low_mean": 0.052641024347394705, "clip_ratio/low_min": 0.052641024347394705, "clip_ratio/high_mean": 0.019731405191123486, "clip_ratio/high_max": 0.019731405191123486, "clip_ratio/region_mean": 0.07237242953851819, "reward_total_mean": 0.8086029291152954, "reward_meter_mean": 0.9734506607055664, "reward_meter_std": 0.019772527739405632, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8855016231536865, "reward_repeat_soft_std": 0.06458554416894913, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.20860078930854797, "reward_total_composite_mean": 0.8086029291152954, "reward_total_composite_std": 0.06107509136199951} {"timestamp_utc": "2026-04-13T04:45:54Z", "mode": "train", "global_step": 2549, "epoch": 0.2560522350577599, "loss": -0.1449, "grad_norm": 1.3421480655670166, "learning_rate": 2.278787878787879e-06, "num_tokens": 4734739.0, "completions/mean_length": 108.875, "completions/min_length": 48.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 51.28571701049805, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9015717506408691, "rewards/meter/std": 0.23000936210155487, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8220525979995728, "rewards/repeat_soft/std": 0.06842949241399765, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.13452960550785065, "rewards/total_composite/mean": 0.7016669511795044, "rewards/total_composite/std": 0.2836363911628723, "reward": 0.7016669511795044, "reward_std": 0.2836363911628723, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05010503903031349, "sampling/sampling_logp_difference/max": 0.8777420520782471, "sampling/importance_sampling_ratio/min": 0.41572052240371704, "sampling/importance_sampling_ratio/mean": 1.0122532844543457, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.28028322383761406, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.04226513113826513, "clip_ratio/high_max": 0.04226513113826513, "clip_ratio/region_mean": 0.04226513113826513, "reward_total_mean": 0.7016669511795044, "reward_meter_mean": 0.9015717506408691, "reward_meter_std": 0.23000936210155487, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8220525979995728, "reward_repeat_soft_std": 0.06842949241399765, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.13452960550785065, "reward_total_composite_mean": 0.7016669511795044, "reward_total_composite_std": 0.2836363911628723} {"timestamp_utc": "2026-04-13T04:46:01Z", "mode": "train", "global_step": 2550, "epoch": 0.25615268709191363, "loss": -0.0118, "grad_norm": 5.9938788414001465, "learning_rate": 2.275757575757576e-06, "num_tokens": 4736773.0, "completions/mean_length": 95.25, "completions/min_length": 86.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.25, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.9929567575454712, "rewards/meter/std": 0.003188579110428691, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9596675038337708, "rewards/repeat_soft/std": 0.01818038709461689, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.81879723072052, "rewards/total_composite/std": 0.002247933764010668, "reward": 0.81879723072052, "reward_std": 0.00224792561493814, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09812991321086884, "sampling/sampling_logp_difference/max": 2.3833816051483154, "sampling/importance_sampling_ratio/min": 0.09223813563585281, "sampling/importance_sampling_ratio/mean": 1.0059266090393066, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5021963678300381, "clip_ratio/low_mean": 0.02404311951249838, "clip_ratio/low_min": 0.02404311951249838, "clip_ratio/high_mean": 0.08659396879374981, "clip_ratio/high_max": 0.08659396879374981, "clip_ratio/region_mean": 0.11063708830624819, "reward_total_mean": 0.81879723072052, "reward_meter_mean": 0.9929567575454712, "reward_meter_std": 0.003188579110428691, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9596675038337708, "reward_repeat_soft_std": 0.01818038709461689, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.81879723072052, "reward_total_composite_std": 0.002247933764010668} {"timestamp_utc": "2026-04-13T04:46:57Z", "mode": "eval", "global_step": 2550, "epoch": 0.25615268709191363, "eval_loss": NaN, "eval_runtime": 56.2634, "eval_samples_per_second": 1.422, "eval_steps_per_second": 0.178, "eval_num_tokens": 4736773.0, "eval_completions/mean_length": 108.375, "eval_completions/min_length": 41.2, "eval_completions/max_length": 241.3, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 98.05357208251954, "eval_completions/min_terminated_length": 41.2, "eval_completions/max_terminated_length": 171.3, "eval_rewards/meter/mean": 0.902036851644516, "eval_rewards/meter/std": 0.17986593283712865, "eval_rewards/count_adherence/mean": 0.9499999821186066, "eval_rewards/count_adherence/std": 0.10748020857572556, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.07071067690849304, "eval_rewards/repeat_soft/mean": 0.8803816556930542, "eval_rewards/repeat_soft/std": 0.11005132459104061, "eval_rewards/judge_quality/mean": 0.42037499845027926, "eval_rewards/judge_quality/std": 0.17186082527041435, "eval_rewards/total_composite/mean": 0.7591127455234528, "eval_rewards/total_composite/std": 0.13903248719871045, "eval_reward": 0.7591127455234528, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.041771961376070976, "eval_sampling/sampling_logp_difference/max": 0.9354195594787598, "eval_sampling/importance_sampling_ratio/min": 0.41063871085643766, "eval_sampling/importance_sampling_ratio/mean": 1.009030854701996, "eval_sampling/importance_sampling_ratio/max": 1.3570029139518738, "eval_entropy": 0.42917735278606417, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7591127455234528, "eval_reward_meter_mean": 0.902036851644516, "eval_reward_meter_std": 0.17986593283712865, "eval_reward_count_adherence_mean": 0.9499999821186066, "eval_reward_count_adherence_std": 0.10748020857572556, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.07071067690849304, "eval_reward_repeat_soft_mean": 0.8803816556930542, "eval_reward_repeat_soft_std": 0.11005132459104061, "eval_reward_judge_quality_mean": 0.42037499845027926, "eval_reward_judge_quality_std": 0.17186082527041435, "eval_reward_total_composite_mean": 0.7591127455234528, "eval_reward_total_composite_std": 0.13903248719871045} {"timestamp_utc": "2026-04-13T04:47:09Z", "mode": "train", "global_step": 2551, "epoch": 0.2562531391260673, "loss": 0.0278, "grad_norm": 14.679176330566406, "learning_rate": 2.2727272727272728e-06, "num_tokens": 4738654.0, "completions/mean_length": 62.125, "completions/min_length": 53.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.125, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.8792790770530701, "rewards/meter/std": 0.2945381999015808, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9180572032928467, "rewards/repeat_soft/std": 0.0721125602722168, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7571063041687012, "rewards/total_composite/std": 0.1472284197807312, "reward": 0.7571063041687012, "reward_std": 0.1472284346818924, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10137269645929337, "sampling/sampling_logp_difference/max": 1.389074444770813, "sampling/importance_sampling_ratio/min": 0.24930594861507416, "sampling/importance_sampling_ratio/mean": 1.0056418180465698, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5718254335224628, "clip_ratio/low_mean": 0.006147541105747223, "clip_ratio/low_min": 0.006147541105747223, "clip_ratio/high_mean": 0.06889402656815946, "clip_ratio/high_max": 0.06889402656815946, "clip_ratio/region_mean": 0.07504156767390668, "reward_total_mean": 0.7571063041687012, "reward_meter_mean": 0.8792790770530701, "reward_meter_std": 0.2945381999015808, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9180572032928467, "reward_repeat_soft_std": 0.0721125602722168, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7571063041687012, "reward_total_composite_std": 0.1472284197807312} {"timestamp_utc": "2026-04-13T04:47:16Z", "mode": "train", "global_step": 2552, "epoch": 0.256353591160221, "loss": 0.018, "grad_norm": 10.652816772460938, "learning_rate": 2.2696969696969696e-06, "num_tokens": 4740462.0, "completions/mean_length": 65.0, "completions/min_length": 62.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9895347952842712, "rewards/meter/std": 0.003434507641941309, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9842222929000854, "rewards/repeat_soft/std": 0.009569668211042881, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.8947128653526306, "rewards/total_composite/std": 0.08061040192842484, "reward": 0.8947128653526306, "reward_std": 0.08061040937900543, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11261554062366486, "sampling/sampling_logp_difference/max": 2.165623664855957, "sampling/importance_sampling_ratio/min": 0.11467839032411575, "sampling/importance_sampling_ratio/mean": 0.9997318983078003, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5786668881773949, "clip_ratio/low_mean": 0.04317499196622521, "clip_ratio/low_min": 0.04317499196622521, "clip_ratio/high_mean": 0.05327662732452154, "clip_ratio/high_max": 0.05327662732452154, "clip_ratio/region_mean": 0.09645161929074675, "reward_total_mean": 0.8947128653526306, "reward_meter_mean": 0.9895347952842712, "reward_meter_std": 0.003434507641941309, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9842222929000854, "reward_repeat_soft_std": 0.009569668211042881, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.8947128653526306, "reward_total_composite_std": 0.08061040192842484} {"timestamp_utc": "2026-04-13T04:47:22Z", "mode": "train", "global_step": 2553, "epoch": 0.2564540431943747, "loss": -0.028, "grad_norm": 6.499220371246338, "learning_rate": 2.266666666666667e-06, "num_tokens": 4742160.0, "completions/mean_length": 49.25, "completions/min_length": 45.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.25, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.8915788531303406, "rewards/meter/std": 0.26911112666130066, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9528446793556213, "rewards/repeat_soft/std": 0.0335821732878685, "rewards/judge_quality/mean": 0.6225000023841858, "rewards/judge_quality/std": 0.24656209349632263, "rewards/total_composite/mean": 0.8332449197769165, "rewards/total_composite/std": 0.10303010791540146, "reward": 0.8332449197769165, "reward_std": 0.10303011536598206, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06491338461637497, "sampling/sampling_logp_difference/max": 0.91225266456604, "sampling/importance_sampling_ratio/min": 0.40161851048469543, "sampling/importance_sampling_ratio/mean": 1.015472650527954, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36269461363554, "clip_ratio/low_mean": 0.06555177667178214, "clip_ratio/low_min": 0.06555177667178214, "clip_ratio/high_mean": 0.004999999888241291, "clip_ratio/high_max": 0.004999999888241291, "clip_ratio/region_mean": 0.07055177656002343, "reward_total_mean": 0.8332449197769165, "reward_meter_mean": 0.8915788531303406, "reward_meter_std": 0.26911112666130066, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9528446793556213, "reward_repeat_soft_std": 0.0335821732878685, "reward_judge_quality_mean": 0.6225000023841858, "reward_judge_quality_std": 0.24656209349632263, "reward_total_composite_mean": 0.8332449197769165, "reward_total_composite_std": 0.10303010791540146} {"timestamp_utc": "2026-04-13T04:47:29Z", "mode": "train", "global_step": 2554, "epoch": 0.25655449522852836, "loss": 0.0338, "grad_norm": 11.612336158752441, "learning_rate": 2.2636363636363637e-06, "num_tokens": 4743846.0, "completions/mean_length": 53.75, "completions/min_length": 49.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.75, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9836316108703613, "rewards/meter/std": 0.006375269498676062, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9906048774719238, "rewards/repeat_soft/std": 0.0109636802226305, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8210697174072266, "rewards/total_composite/std": 0.005021795630455017, "reward": 0.8210697174072266, "reward_std": 0.005021798424422741, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12846535444259644, "sampling/sampling_logp_difference/max": 2.102322578430176, "sampling/importance_sampling_ratio/min": 0.12217234820127487, "sampling/importance_sampling_ratio/mean": 0.994185209274292, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6097560413181782, "clip_ratio/low_mean": 0.04955746000632644, "clip_ratio/low_min": 0.04955746000632644, "clip_ratio/high_mean": 0.06279828539118171, "clip_ratio/high_max": 0.06279828539118171, "clip_ratio/region_mean": 0.11235574539750814, "reward_total_mean": 0.8210697174072266, "reward_meter_mean": 0.9836316108703613, "reward_meter_std": 0.006375269498676062, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9906048774719238, "reward_repeat_soft_std": 0.0109636802226305, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8210697174072266, "reward_total_composite_std": 0.005021795630455017} {"timestamp_utc": "2026-04-13T04:47:39Z", "mode": "train", "global_step": 2555, "epoch": 0.25665494726268206, "loss": -0.0221, "grad_norm": 4.271241188049316, "learning_rate": 2.260606060606061e-06, "num_tokens": 4746404.0, "completions/mean_length": 139.75, "completions/min_length": 128.0, "completions/max_length": 150.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 139.75, "completions/min_terminated_length": 128.0, "completions/max_terminated_length": 150.0, "rewards/meter/mean": 0.9913886785507202, "rewards/meter/std": 0.005592599045485258, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6717339158058167, "rewards/repeat_soft/std": 0.08023279905319214, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7765482664108276, "rewards/total_composite/std": 0.023932231590151787, "reward": 0.7765482664108276, "reward_std": 0.02393222786486149, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07387859374284744, "sampling/sampling_logp_difference/max": 1.7540373802185059, "sampling/importance_sampling_ratio/min": 0.17307376861572266, "sampling/importance_sampling_ratio/mean": 1.0163462162017822, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37893347814679146, "clip_ratio/low_mean": 0.021153959445655346, "clip_ratio/low_min": 0.021153959445655346, "clip_ratio/high_mean": 0.053514239843934774, "clip_ratio/high_max": 0.053514239843934774, "clip_ratio/region_mean": 0.07466819928959012, "reward_total_mean": 0.7765482664108276, "reward_meter_mean": 0.9913886785507202, "reward_meter_std": 0.005592599045485258, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6717339158058167, "reward_repeat_soft_std": 0.08023279905319214, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7765482664108276, "reward_total_composite_std": 0.023932231590151787} {"timestamp_utc": "2026-04-13T04:47:45Z", "mode": "train", "global_step": 2556, "epoch": 0.2567553992968358, "loss": -0.0507, "grad_norm": 14.867445945739746, "learning_rate": 2.2575757575757578e-06, "num_tokens": 4747773.0, "completions/mean_length": 29.125, "completions/min_length": 25.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.125, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.8509201407432556, "rewards/meter/std": 0.2561531960964203, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9513168334960938, "rewards/repeat_soft/std": 0.017598045989871025, "rewards/judge_quality/mean": 0.8025000095367432, "rewards/judge_quality/std": 0.21756774187088013, "rewards/total_composite/mean": 0.8687957525253296, "rewards/total_composite/std": 0.1495727151632309, "reward": 0.8687957525253296, "reward_std": 0.1495727002620697, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12028888612985611, "sampling/sampling_logp_difference/max": 1.7856621742248535, "sampling/importance_sampling_ratio/min": 0.16768598556518555, "sampling/importance_sampling_ratio/mean": 0.9879758358001709, "sampling/importance_sampling_ratio/max": 1.6447919607162476, "entropy": 0.7364695742726326, "clip_ratio/low_mean": 0.042817460373044014, "clip_ratio/low_min": 0.042817460373044014, "clip_ratio/high_mean": 0.07603240245953202, "clip_ratio/high_max": 0.07603240245953202, "clip_ratio/region_mean": 0.11884986283257604, "reward_total_mean": 0.8687957525253296, "reward_meter_mean": 0.8509201407432556, "reward_meter_std": 0.2561531960964203, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9513168334960938, "reward_repeat_soft_std": 0.017598045989871025, "reward_judge_quality_mean": 0.8025000095367432, "reward_judge_quality_std": 0.21756774187088013, "reward_total_composite_mean": 0.8687957525253296, "reward_total_composite_std": 0.1495727151632309} {"timestamp_utc": "2026-04-13T04:47:53Z", "mode": "train", "global_step": 2557, "epoch": 0.2568558513309894, "loss": 0.0156, "grad_norm": 6.180031776428223, "learning_rate": 2.254545454545455e-06, "num_tokens": 4750072.0, "completions/mean_length": 117.375, "completions/min_length": 109.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.375, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.9841905832290649, "rewards/meter/std": 0.012659740634262562, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9223132133483887, "rewards/repeat_soft/std": 0.04329851269721985, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8111170530319214, "rewards/total_composite/std": 0.0066663287580013275, "reward": 0.8111170530319214, "reward_std": 0.0066663287580013275, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08405380696058273, "sampling/sampling_logp_difference/max": 1.5606536865234375, "sampling/importance_sampling_ratio/min": 0.20999875664710999, "sampling/importance_sampling_ratio/mean": 1.0027459859848022, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39664334803819656, "clip_ratio/low_mean": 0.022094298619776964, "clip_ratio/low_min": 0.022094298619776964, "clip_ratio/high_mean": 0.053455686662346125, "clip_ratio/high_max": 0.053455686662346125, "clip_ratio/region_mean": 0.07554998528212309, "reward_total_mean": 0.8111170530319214, "reward_meter_mean": 0.9841905832290649, "reward_meter_std": 0.012659740634262562, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9223132133483887, "reward_repeat_soft_std": 0.04329851269721985, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8111170530319214, "reward_total_composite_std": 0.0066663287580013275} {"timestamp_utc": "2026-04-13T04:48:00Z", "mode": "train", "global_step": 2558, "epoch": 0.25695630336514313, "loss": 0.0025, "grad_norm": 11.541792869567871, "learning_rate": 2.251515151515152e-06, "num_tokens": 4751580.0, "completions/mean_length": 30.5, "completions/min_length": 27.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.5, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.8161038160324097, "rewards/meter/std": 0.33339086174964905, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9600961208343506, "rewards/repeat_soft/std": 0.006799087394028902, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7415063381195068, "rewards/total_composite/std": 0.14586050808429718, "reward": 0.7415063381195068, "reward_std": 0.145860493183136, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0769181028008461, "sampling/sampling_logp_difference/max": 1.796346664428711, "sampling/importance_sampling_ratio/min": 0.16590388119220734, "sampling/importance_sampling_ratio/mean": 1.0003070831298828, "sampling/importance_sampling_ratio/max": 1.6266446113586426, "entropy": 0.3611031174659729, "clip_ratio/low_mean": 0.016183035913854837, "clip_ratio/low_min": 0.016183035913854837, "clip_ratio/high_mean": 0.06861745798960328, "clip_ratio/high_max": 0.06861745798960328, "clip_ratio/region_mean": 0.08480049390345812, "reward_total_mean": 0.7415063381195068, "reward_meter_mean": 0.8161038160324097, "reward_meter_std": 0.33339086174964905, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9600961208343506, "reward_repeat_soft_std": 0.006799087394028902, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7415063381195068, "reward_total_composite_std": 0.14586050808429718} {"timestamp_utc": "2026-04-13T04:48:12Z", "mode": "train", "global_step": 2559, "epoch": 0.25705675539929684, "loss": -0.1826, "grad_norm": 1.47858726978302, "learning_rate": 2.2484848484848487e-06, "num_tokens": 4753529.0, "completions/mean_length": 195.625, "completions/min_length": 78.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 90.16667175292969, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.9631693363189697, "rewards/meter/std": 0.0732659325003624, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.8076298236846924, "rewards/repeat_soft/std": 0.06757467985153198, "rewards/judge_quality/mean": 0.2887499928474426, "rewards/judge_quality/std": 0.20517851412296295, "rewards/total_composite/mean": 0.5851422548294067, "rewards/total_composite/std": 0.3643777370452881, "reward": 0.5851422548294067, "reward_std": 0.3643777370452881, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09748553484678268, "sampling/sampling_logp_difference/max": 1.7304902076721191, "sampling/importance_sampling_ratio/min": 0.17719751596450806, "sampling/importance_sampling_ratio/mean": 0.9999764561653137, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4019298255443573, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08365907799452543, "clip_ratio/high_max": 0.08365907799452543, "clip_ratio/region_mean": 0.08365907799452543, "reward_total_mean": 0.5851422548294067, "reward_meter_mean": 0.9631693363189697, "reward_meter_std": 0.0732659325003624, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.8076298236846924, "reward_repeat_soft_std": 0.06757467985153198, "reward_judge_quality_mean": 0.2887499928474426, "reward_judge_quality_std": 0.20517851412296295, "reward_total_composite_mean": 0.5851422548294067, "reward_total_composite_std": 0.3643777370452881} {"timestamp_utc": "2026-04-13T04:48:20Z", "mode": "train", "global_step": 2560, "epoch": 0.25715720743345055, "loss": -0.002, "grad_norm": 5.627600193023682, "learning_rate": 2.2454545454545455e-06, "num_tokens": 4755913.0, "completions/mean_length": 124.0, "completions/min_length": 115.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 124.0, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.9877270460128784, "rewards/meter/std": 0.003347694408148527, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7985920906066895, "rewards/repeat_soft/std": 0.0929899662733078, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8003363609313965, "rewards/total_composite/std": 0.009253918193280697, "reward": 0.8003363609313965, "reward_std": 0.009253905154764652, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07864027470350266, "sampling/sampling_logp_difference/max": 1.890045166015625, "sampling/importance_sampling_ratio/min": 0.15106497704982758, "sampling/importance_sampling_ratio/mean": 1.0089795589447021, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3997655101120472, "clip_ratio/low_mean": 0.029688720125705004, "clip_ratio/low_min": 0.029688720125705004, "clip_ratio/high_mean": 0.05166061548516154, "clip_ratio/high_max": 0.05166061548516154, "clip_ratio/region_mean": 0.08134933561086655, "reward_total_mean": 0.8003363609313965, "reward_meter_mean": 0.9877270460128784, "reward_meter_std": 0.003347694408148527, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7985920906066895, "reward_repeat_soft_std": 0.0929899662733078, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8003363609313965, "reward_total_composite_std": 0.009253918193280697} {"timestamp_utc": "2026-04-13T04:48:27Z", "mode": "train", "global_step": 2561, "epoch": 0.2572576594676042, "loss": 0.1439, "grad_norm": 7.308074474334717, "learning_rate": 2.2424242424242428e-06, "num_tokens": 4757565.0, "completions/mean_length": 49.5, "completions/min_length": 45.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.5, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9620589017868042, "rewards/meter/std": 0.008773819543421268, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9017727375030518, "rewards/repeat_soft/std": 0.07982788234949112, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.793103814125061, "rewards/total_composite/std": 0.032847821712493896, "reward": 0.793103814125061, "reward_std": 0.03284782916307449, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0628509446978569, "sampling/sampling_logp_difference/max": 1.3508415222167969, "sampling/importance_sampling_ratio/min": 0.25902220606803894, "sampling/importance_sampling_ratio/mean": 1.006556510925293, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3195287436246872, "clip_ratio/low_mean": 0.017460317816585302, "clip_ratio/low_min": 0.017460317816585302, "clip_ratio/high_mean": 0.039541025878861547, "clip_ratio/high_max": 0.039541025878861547, "clip_ratio/region_mean": 0.05700134369544685, "reward_total_mean": 0.793103814125061, "reward_meter_mean": 0.9620589017868042, "reward_meter_std": 0.008773819543421268, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9017727375030518, "reward_repeat_soft_std": 0.07982788234949112, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.793103814125061, "reward_total_composite_std": 0.032847821712493896} {"timestamp_utc": "2026-04-13T04:48:35Z", "mode": "train", "global_step": 2562, "epoch": 0.2573581115017579, "loss": 0.0399, "grad_norm": 9.015654563903809, "learning_rate": 2.2393939393939396e-06, "num_tokens": 4759675.0, "completions/mean_length": 85.75, "completions/min_length": 77.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.75, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.9659424424171448, "rewards/meter/std": 0.038848329335451126, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.894291877746582, "rewards/repeat_soft/std": 0.054220154881477356, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.8047907948493958, "rewards/total_composite/std": 0.0665271133184433, "reward": 0.8047907948493958, "reward_std": 0.0665271207690239, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09530884772539139, "sampling/sampling_logp_difference/max": 2.966644287109375, "sampling/importance_sampling_ratio/min": 0.05147575959563255, "sampling/importance_sampling_ratio/mean": 0.9879980087280273, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4289683401584625, "clip_ratio/low_mean": 0.06499910913407803, "clip_ratio/low_min": 0.06499910913407803, "clip_ratio/high_mean": 0.023646533489227295, "clip_ratio/high_max": 0.023646533489227295, "clip_ratio/region_mean": 0.08864564262330532, "reward_total_mean": 0.8047907948493958, "reward_meter_mean": 0.9659424424171448, "reward_meter_std": 0.038848329335451126, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.894291877746582, "reward_repeat_soft_std": 0.054220154881477356, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.8047907948493958, "reward_total_composite_std": 0.0665271133184433} {"timestamp_utc": "2026-04-13T04:48:42Z", "mode": "train", "global_step": 2563, "epoch": 0.2574585635359116, "loss": 0.0059, "grad_norm": 10.591803550720215, "learning_rate": 2.2363636363636364e-06, "num_tokens": 4761399.0, "completions/mean_length": 60.5, "completions/min_length": 56.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.5, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9934835433959961, "rewards/meter/std": 0.006659524515271187, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9565032720565796, "rewards/repeat_soft/std": 0.03914221376180649, "rewards/judge_quality/mean": 0.5012500286102295, "rewards/judge_quality/std": 0.16974246501922607, "rewards/total_composite/mean": 0.8430929183959961, "rewards/total_composite/std": 0.04853714630007744, "reward": 0.8430929183959961, "reward_std": 0.04853714630007744, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09060289710760117, "sampling/sampling_logp_difference/max": 1.2222659587860107, "sampling/importance_sampling_ratio/min": 0.2945619523525238, "sampling/importance_sampling_ratio/mean": 1.0037775039672852, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48457475006580353, "clip_ratio/low_mean": 0.09556638775393367, "clip_ratio/low_min": 0.09556638775393367, "clip_ratio/high_mean": 0.007936508394777775, "clip_ratio/high_max": 0.007936508394777775, "clip_ratio/region_mean": 0.10350289614871144, "reward_total_mean": 0.8430929183959961, "reward_meter_mean": 0.9934835433959961, "reward_meter_std": 0.006659524515271187, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9565032720565796, "reward_repeat_soft_std": 0.03914221376180649, "reward_judge_quality_mean": 0.5012500286102295, "reward_judge_quality_std": 0.16974246501922607, "reward_total_composite_mean": 0.8430929183959961, "reward_total_composite_std": 0.04853714630007744} {"timestamp_utc": "2026-04-13T04:48:49Z", "mode": "train", "global_step": 2564, "epoch": 0.2575590155700653, "loss": -0.009, "grad_norm": 9.80305290222168, "learning_rate": 2.2333333333333333e-06, "num_tokens": 4763151.0, "completions/mean_length": 58.0, "completions/min_length": 53.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9391530752182007, "rewards/meter/std": 0.13334892690181732, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9579945802688599, "rewards/repeat_soft/std": 0.027897851541638374, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7955433130264282, "rewards/total_composite/std": 0.061657536774873734, "reward": 0.7955433130264282, "reward_std": 0.061657536774873734, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09433142840862274, "sampling/sampling_logp_difference/max": 1.5746151208877563, "sampling/importance_sampling_ratio/min": 0.20708723366260529, "sampling/importance_sampling_ratio/mean": 1.000282883644104, "sampling/importance_sampling_ratio/max": 1.9692596197128296, "entropy": 0.4021543189883232, "clip_ratio/low_mean": 0.004716981202363968, "clip_ratio/low_min": 0.004716981202363968, "clip_ratio/high_mean": 0.07870091823861003, "clip_ratio/high_max": 0.07870091823861003, "clip_ratio/region_mean": 0.083417899440974, "reward_total_mean": 0.7955433130264282, "reward_meter_mean": 0.9391530752182007, "reward_meter_std": 0.13334892690181732, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9579945802688599, "reward_repeat_soft_std": 0.027897851541638374, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7955433130264282, "reward_total_composite_std": 0.061657536774873734} {"timestamp_utc": "2026-04-13T04:48:57Z", "mode": "train", "global_step": 2565, "epoch": 0.257659467604219, "loss": 0.0592, "grad_norm": 6.9192328453063965, "learning_rate": 2.2303030303030305e-06, "num_tokens": 4765683.0, "completions/mean_length": 131.5, "completions/min_length": 100.0, "completions/max_length": 150.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.5, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 150.0, "rewards/meter/mean": 0.8137656450271606, "rewards/meter/std": 0.2953343093395233, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.07715168595314026, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6215099096298218, "rewards/repeat_soft/std": 0.19600659608840942, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.6600955724716187, "rewards/total_composite/std": 0.12183001637458801, "reward": 0.6600955724716187, "reward_std": 0.12183001637458801, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07440313696861267, "sampling/sampling_logp_difference/max": 1.7316515445709229, "sampling/importance_sampling_ratio/min": 0.17699186503887177, "sampling/importance_sampling_ratio/mean": 0.9950737953186035, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2920130733400583, "clip_ratio/low_mean": 0.010106837842613459, "clip_ratio/low_min": 0.010106837842613459, "clip_ratio/high_mean": 0.05048635113053024, "clip_ratio/high_max": 0.05048635113053024, "clip_ratio/region_mean": 0.0605931889731437, "reward_total_mean": 0.6600955724716187, "reward_meter_mean": 0.8137656450271606, "reward_meter_std": 0.2953343093395233, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.07715168595314026, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6215099096298218, "reward_repeat_soft_std": 0.19600659608840942, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.6600955724716187, "reward_total_composite_std": 0.12183001637458801} {"timestamp_utc": "2026-04-13T04:49:09Z", "mode": "train", "global_step": 2566, "epoch": 0.2577599196383727, "loss": -0.1347, "grad_norm": 1.7241398096084595, "learning_rate": 2.2272727272727274e-06, "num_tokens": 4767418.0, "completions/mean_length": 239.875, "completions/min_length": 70.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 76.5999984741211, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.6101977229118347, "rewards/meter/std": 0.4258233606815338, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9011238813400269, "rewards/repeat_soft/std": 0.13447776436805725, "rewards/judge_quality/mean": 0.3400000035762787, "rewards/judge_quality/std": 0.2789265513420105, "rewards/total_composite/mean": 0.49162524938583374, "rewards/total_composite/std": 0.41652432084083557, "reward": 0.49162524938583374, "reward_std": 0.41652432084083557, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12401630729436874, "sampling/sampling_logp_difference/max": 1.992330551147461, "sampling/importance_sampling_ratio/min": 0.136377215385437, "sampling/importance_sampling_ratio/mean": 1.002335548400879, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3708145171403885, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0783514566719532, "clip_ratio/high_max": 0.0783514566719532, "clip_ratio/region_mean": 0.0783514566719532, "reward_total_mean": 0.49162524938583374, "reward_meter_mean": 0.6101977229118347, "reward_meter_std": 0.4258233606815338, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9011238813400269, "reward_repeat_soft_std": 0.13447776436805725, "reward_judge_quality_mean": 0.3400000035762787, "reward_judge_quality_std": 0.2789265513420105, "reward_total_composite_mean": 0.49162524938583374, "reward_total_composite_std": 0.41652432084083557} {"timestamp_utc": "2026-04-13T04:49:17Z", "mode": "train", "global_step": 2567, "epoch": 0.25786037167252635, "loss": 0.0168, "grad_norm": 6.14030647277832, "learning_rate": 2.224242424242424e-06, "num_tokens": 4769952.0, "completions/mean_length": 140.75, "completions/min_length": 134.0, "completions/max_length": 150.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 140.75, "completions/min_terminated_length": 134.0, "completions/max_terminated_length": 150.0, "rewards/meter/mean": 0.9925752878189087, "rewards/meter/std": 0.002189001766964793, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9500036835670471, "rewards/repeat_soft/std": 0.018658574670553207, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7985342741012573, "rewards/total_composite/std": 0.025585193186998367, "reward": 0.7985342741012573, "reward_std": 0.025585198774933815, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10218992084264755, "sampling/sampling_logp_difference/max": 2.8692591190338135, "sampling/importance_sampling_ratio/min": 0.05674095079302788, "sampling/importance_sampling_ratio/mean": 0.9997554421424866, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5366576500236988, "clip_ratio/low_mean": 0.024849604349583387, "clip_ratio/low_min": 0.024849604349583387, "clip_ratio/high_mean": 0.07591282669454813, "clip_ratio/high_max": 0.07591282669454813, "clip_ratio/region_mean": 0.10076243104413152, "reward_total_mean": 0.7985342741012573, "reward_meter_mean": 0.9925752878189087, "reward_meter_std": 0.002189001766964793, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9500036835670471, "reward_repeat_soft_std": 0.018658574670553207, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7985342741012573, "reward_total_composite_std": 0.025585193186998367} {"timestamp_utc": "2026-04-13T04:49:25Z", "mode": "train", "global_step": 2568, "epoch": 0.25796082370668005, "loss": 0.0708, "grad_norm": 11.13090991973877, "learning_rate": 2.2212121212121214e-06, "num_tokens": 4771679.0, "completions/mean_length": 48.875, "completions/min_length": 38.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.875, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.6593063473701477, "rewards/meter/std": 0.35222676396369934, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9771648049354553, "rewards/repeat_soft/std": 0.018377162516117096, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.6640293598175049, "rewards/total_composite/std": 0.1648591309785843, "reward": 0.6640293598175049, "reward_std": 0.1648591160774231, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10790430009365082, "sampling/sampling_logp_difference/max": 1.3175086975097656, "sampling/importance_sampling_ratio/min": 0.2678016424179077, "sampling/importance_sampling_ratio/mean": 1.0234555006027222, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5537565425038338, "clip_ratio/low_mean": 0.05076410621404648, "clip_ratio/low_min": 0.05076410621404648, "clip_ratio/high_mean": 0.04793610330671072, "clip_ratio/high_max": 0.04793610330671072, "clip_ratio/region_mean": 0.0987002095207572, "reward_total_mean": 0.6640293598175049, "reward_meter_mean": 0.6593063473701477, "reward_meter_std": 0.35222676396369934, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9771648049354553, "reward_repeat_soft_std": 0.018377162516117096, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.6640293598175049, "reward_total_composite_std": 0.1648591309785843} {"timestamp_utc": "2026-04-13T04:49:32Z", "mode": "train", "global_step": 2569, "epoch": 0.25806127574083376, "loss": 0.0213, "grad_norm": 7.951539993286133, "learning_rate": 2.2181818181818187e-06, "num_tokens": 4773499.0, "completions/mean_length": 61.5, "completions/min_length": 52.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.5, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9880010485649109, "rewards/meter/std": 0.010813278146088123, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9085690975189209, "rewards/repeat_soft/std": 0.11023641377687454, "rewards/judge_quality/mean": 0.5349999666213989, "rewards/judge_quality/std": 0.18431341648101807, "rewards/total_composite/mean": 0.8459573984146118, "rewards/total_composite/std": 0.06007236987352371, "reward": 0.8459573984146118, "reward_std": 0.06007235869765282, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09987130016088486, "sampling/sampling_logp_difference/max": 2.104982376098633, "sampling/importance_sampling_ratio/min": 0.12184781581163406, "sampling/importance_sampling_ratio/mean": 1.0018658638000488, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4422670193016529, "clip_ratio/low_mean": 0.07288530841469765, "clip_ratio/low_min": 0.07288530841469765, "clip_ratio/high_mean": 0.024603175930678844, "clip_ratio/high_max": 0.024603175930678844, "clip_ratio/region_mean": 0.09748848434537649, "reward_total_mean": 0.8459573984146118, "reward_meter_mean": 0.9880010485649109, "reward_meter_std": 0.010813278146088123, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9085690975189209, "reward_repeat_soft_std": 0.11023641377687454, "reward_judge_quality_mean": 0.5349999666213989, "reward_judge_quality_std": 0.18431341648101807, "reward_total_composite_mean": 0.8459573984146118, "reward_total_composite_std": 0.06007236987352371} {"timestamp_utc": "2026-04-13T04:49:39Z", "mode": "train", "global_step": 2570, "epoch": 0.25816172777498747, "loss": 0.0372, "grad_norm": 21.068586349487305, "learning_rate": 2.2151515151515155e-06, "num_tokens": 4775174.0, "completions/mean_length": 38.375, "completions/min_length": 36.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.375, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.8844739198684692, "rewards/meter/std": 0.27837714552879333, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9141753911972046, "rewards/repeat_soft/std": 0.06866205483675003, "rewards/judge_quality/mean": 0.41749998927116394, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.764680802822113, "rewards/total_composite/std": 0.14317145943641663, "reward": 0.764680802822113, "reward_std": 0.14317144453525543, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1232629045844078, "sampling/sampling_logp_difference/max": 2.2099828720092773, "sampling/importance_sampling_ratio/min": 0.10970252752304077, "sampling/importance_sampling_ratio/mean": 1.0077323913574219, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5340330377221107, "clip_ratio/low_mean": 0.016025641933083534, "clip_ratio/low_min": 0.016025641933083534, "clip_ratio/high_mean": 0.09049495030194521, "clip_ratio/high_max": 0.09049495030194521, "clip_ratio/region_mean": 0.10652059223502874, "reward_total_mean": 0.764680802822113, "reward_meter_mean": 0.8844739198684692, "reward_meter_std": 0.27837714552879333, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9141753911972046, "reward_repeat_soft_std": 0.06866205483675003, "reward_judge_quality_mean": 0.41749998927116394, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.764680802822113, "reward_total_composite_std": 0.14317145943641663} {"timestamp_utc": "2026-04-13T04:49:46Z", "mode": "train", "global_step": 2571, "epoch": 0.2582621798091411, "loss": 0.0465, "grad_norm": 8.458503723144531, "learning_rate": 2.2121212121212124e-06, "num_tokens": 4777177.0, "completions/mean_length": 89.375, "completions/min_length": 86.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.375, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.9865642786026001, "rewards/meter/std": 0.0033174706622958183, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8993933796882629, "rewards/repeat_soft/std": 0.05707050487399101, "rewards/judge_quality/mean": 0.26375001668930054, "rewards/judge_quality/std": 0.13373079895973206, "rewards/total_composite/mean": 0.7630182504653931, "rewards/total_composite/std": 0.04396125674247742, "reward": 0.7630182504653931, "reward_std": 0.04396125674247742, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09153833985328674, "sampling/sampling_logp_difference/max": 1.174649715423584, "sampling/importance_sampling_ratio/min": 0.30892717838287354, "sampling/importance_sampling_ratio/mean": 1.0021880865097046, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.513318482786417, "clip_ratio/low_mean": 0.044107154943048954, "clip_ratio/low_min": 0.044107154943048954, "clip_ratio/high_mean": 0.05038215406239033, "clip_ratio/high_max": 0.05038215406239033, "clip_ratio/region_mean": 0.09448930900543928, "reward_total_mean": 0.7630182504653931, "reward_meter_mean": 0.9865642786026001, "reward_meter_std": 0.0033174706622958183, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8993933796882629, "reward_repeat_soft_std": 0.05707050487399101, "reward_judge_quality_mean": 0.26375001668930054, "reward_judge_quality_std": 0.13373079895973206, "reward_total_composite_mean": 0.7630182504653931, "reward_total_composite_std": 0.04396125674247742} {"timestamp_utc": "2026-04-13T04:49:53Z", "mode": "train", "global_step": 2572, "epoch": 0.25836263184329483, "loss": -0.0027, "grad_norm": 5.966355800628662, "learning_rate": 2.209090909090909e-06, "num_tokens": 4779612.0, "completions/mean_length": 127.375, "completions/min_length": 108.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.375, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.972981333732605, "rewards/meter/std": 0.03849993273615837, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9410327672958374, "rewards/repeat_soft/std": 0.0286888275295496, "rewards/judge_quality/mean": 0.3137499988079071, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7760698795318604, "rewards/total_composite/std": 0.03832376375794411, "reward": 0.7760698795318604, "reward_std": 0.038323771208524704, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09600324183702469, "sampling/sampling_logp_difference/max": 2.482522964477539, "sampling/importance_sampling_ratio/min": 0.08353220671415329, "sampling/importance_sampling_ratio/mean": 1.0048766136169434, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5004251897335052, "clip_ratio/low_mean": 0.0656418465077877, "clip_ratio/low_min": 0.0656418465077877, "clip_ratio/high_mean": 0.02827848168089986, "clip_ratio/high_max": 0.02827848168089986, "clip_ratio/region_mean": 0.09392032818868756, "reward_total_mean": 0.7760698795318604, "reward_meter_mean": 0.972981333732605, "reward_meter_std": 0.03849993273615837, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9410327672958374, "reward_repeat_soft_std": 0.0286888275295496, "reward_judge_quality_mean": 0.3137499988079071, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7760698795318604, "reward_total_composite_std": 0.03832376375794411} {"timestamp_utc": "2026-04-13T04:49:59Z", "mode": "train", "global_step": 2573, "epoch": 0.25846308387744854, "loss": -0.0292, "grad_norm": 7.546812534332275, "learning_rate": 2.2060606060606064e-06, "num_tokens": 4781350.0, "completions/mean_length": 52.25, "completions/min_length": 48.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.25, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.835828423500061, "rewards/meter/std": 0.2508355677127838, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9084338545799255, "rewards/repeat_soft/std": 0.03705362603068352, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.22403763234615326, "rewards/total_composite/mean": 0.7452161312103271, "rewards/total_composite/std": 0.06579412519931793, "reward": 0.7452161312103271, "reward_std": 0.06579412519931793, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06547343730926514, "sampling/sampling_logp_difference/max": 1.175821304321289, "sampling/importance_sampling_ratio/min": 0.3085654377937317, "sampling/importance_sampling_ratio/mean": 1.0035426616668701, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3359949290752411, "clip_ratio/low_mean": 0.019914216361939907, "clip_ratio/low_min": 0.019914216361939907, "clip_ratio/high_mean": 0.03953468007966876, "clip_ratio/high_max": 0.03953468007966876, "clip_ratio/region_mean": 0.05944889644160867, "reward_total_mean": 0.7452161312103271, "reward_meter_mean": 0.835828423500061, "reward_meter_std": 0.2508355677127838, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9084338545799255, "reward_repeat_soft_std": 0.03705362603068352, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.22403763234615326, "reward_total_composite_mean": 0.7452161312103271, "reward_total_composite_std": 0.06579412519931793} {"timestamp_utc": "2026-04-13T04:50:05Z", "mode": "train", "global_step": 2574, "epoch": 0.2585635359116022, "loss": 0.0622, "grad_norm": 12.926291465759277, "learning_rate": 2.2030303030303033e-06, "num_tokens": 4782848.0, "completions/mean_length": 27.25, "completions/min_length": 25.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.25, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.9184542894363403, "rewards/meter/std": 0.16999875009059906, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4124999940395355, "rewards/judge_quality/std": 0.1060660108923912, "rewards/total_composite/mean": 0.7833044528961182, "rewards/total_composite/std": 0.07788022607564926, "reward": 0.7833044528961182, "reward_std": 0.07788024097681046, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09395311027765274, "sampling/sampling_logp_difference/max": 2.6034328937530518, "sampling/importance_sampling_ratio/min": 0.07401904463768005, "sampling/importance_sampling_ratio/mean": 1.0014907121658325, "sampling/importance_sampling_ratio/max": 1.9119038581848145, "entropy": 0.444498173892498, "clip_ratio/low_mean": 0.016964286100119352, "clip_ratio/low_min": 0.016964286100119352, "clip_ratio/high_mean": 0.05639541568234563, "clip_ratio/high_max": 0.05639541568234563, "clip_ratio/region_mean": 0.07335970178246498, "reward_total_mean": 0.7833044528961182, "reward_meter_mean": 0.9184542894363403, "reward_meter_std": 0.16999875009059906, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4124999940395355, "reward_judge_quality_std": 0.1060660108923912, "reward_total_composite_mean": 0.7833044528961182, "reward_total_composite_std": 0.07788022607564926} {"timestamp_utc": "2026-04-13T04:50:11Z", "mode": "train", "global_step": 2575, "epoch": 0.2586639879457559, "loss": 0.0172, "grad_norm": 6.20088529586792, "learning_rate": 2.2e-06, "num_tokens": 4784838.0, "completions/mean_length": 84.75, "completions/min_length": 79.0, "completions/max_length": 89.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.75, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 89.0, "rewards/meter/mean": 0.9799001216888428, "rewards/meter/std": 0.019697966054081917, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7640202045440674, "rewards/repeat_soft/std": 0.06194280460476875, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.7881070971488953, "rewards/total_composite/std": 0.0221012644469738, "reward": 0.7881070971488953, "reward_std": 0.022101256996393204, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08774153888225555, "sampling/sampling_logp_difference/max": 1.4587249755859375, "sampling/importance_sampling_ratio/min": 0.2325325757265091, "sampling/importance_sampling_ratio/mean": 0.9996246099472046, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3586681578308344, "clip_ratio/low_mean": 0.022889914456754923, "clip_ratio/low_min": 0.022889914456754923, "clip_ratio/high_mean": 0.06729580601677299, "clip_ratio/high_max": 0.06729580601677299, "clip_ratio/region_mean": 0.09018572047352791, "reward_total_mean": 0.7881070971488953, "reward_meter_mean": 0.9799001216888428, "reward_meter_std": 0.019697966054081917, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7640202045440674, "reward_repeat_soft_std": 0.06194280460476875, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.7881070971488953, "reward_total_composite_std": 0.0221012644469738} {"timestamp_utc": "2026-04-13T04:50:18Z", "mode": "train", "global_step": 2576, "epoch": 0.2587644399799096, "loss": 0.0399, "grad_norm": 10.556485176086426, "learning_rate": 2.196969696969697e-06, "num_tokens": 4786575.0, "completions/mean_length": 44.125, "completions/min_length": 41.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.125, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9367766380310059, "rewards/meter/std": 0.03166547790169716, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8807536363601685, "rewards/repeat_soft/std": 0.1626870036125183, "rewards/judge_quality/mean": 0.3762499988079071, "rewards/judge_quality/std": 0.11287634819746017, "rewards/total_composite/mean": 0.7724998593330383, "rewards/total_composite/std": 0.04772472009062767, "reward": 0.7724998593330383, "reward_std": 0.047724734991788864, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09488740563392639, "sampling/sampling_logp_difference/max": 1.4620904922485352, "sampling/importance_sampling_ratio/min": 0.23175130784511566, "sampling/importance_sampling_ratio/mean": 1.0119210481643677, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49788231402635574, "clip_ratio/low_mean": 0.04207091499119997, "clip_ratio/low_min": 0.04207091499119997, "clip_ratio/high_mean": 0.05471561290323734, "clip_ratio/high_max": 0.05471561290323734, "clip_ratio/region_mean": 0.09678652789443731, "reward_total_mean": 0.7724998593330383, "reward_meter_mean": 0.9367766380310059, "reward_meter_std": 0.03166547790169716, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8807536363601685, "reward_repeat_soft_std": 0.1626870036125183, "reward_judge_quality_mean": 0.3762499988079071, "reward_judge_quality_std": 0.11287634819746017, "reward_total_composite_mean": 0.7724998593330383, "reward_total_composite_std": 0.04772472009062767} {"timestamp_utc": "2026-04-13T04:50:25Z", "mode": "train", "global_step": 2577, "epoch": 0.25886489201406326, "loss": 0.0196, "grad_norm": 12.908655166625977, "learning_rate": 2.193939393939394e-06, "num_tokens": 4788162.0, "completions/mean_length": 33.375, "completions/min_length": 29.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.375, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9780943393707275, "rewards/meter/std": 0.01913829706609249, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.5674999952316284, "rewards/judge_quality/std": 0.21756774187088013, "rewards/total_composite/mean": 0.8566424250602722, "rewards/total_composite/std": 0.060389865189790726, "reward": 0.8566424250602722, "reward_std": 0.060389865189790726, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1034965068101883, "sampling/sampling_logp_difference/max": 1.3477191925048828, "sampling/importance_sampling_ratio/min": 0.2598322033882141, "sampling/importance_sampling_ratio/mean": 1.022377610206604, "sampling/importance_sampling_ratio/max": 1.7098963260650635, "entropy": 0.5327597334980965, "clip_ratio/low_mean": 0.09812735626474023, "clip_ratio/low_min": 0.09812735626474023, "clip_ratio/high_mean": 0.026292335707694292, "clip_ratio/high_max": 0.026292335707694292, "clip_ratio/region_mean": 0.12441969197243452, "reward_total_mean": 0.8566424250602722, "reward_meter_mean": 0.9780943393707275, "reward_meter_std": 0.01913829706609249, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.5674999952316284, "reward_judge_quality_std": 0.21756774187088013, "reward_total_composite_mean": 0.8566424250602722, "reward_total_composite_std": 0.060389865189790726} {"timestamp_utc": "2026-04-13T04:50:32Z", "mode": "train", "global_step": 2578, "epoch": 0.25896534404821697, "loss": 0.0152, "grad_norm": 7.053152084350586, "learning_rate": 2.190909090909091e-06, "num_tokens": 4790371.0, "completions/mean_length": 83.125, "completions/min_length": 71.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.125, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.9816865921020508, "rewards/meter/std": 0.014967069961130619, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9673194289207458, "rewards/repeat_soft/std": 0.03114643134176731, "rewards/judge_quality/mean": 0.45250001549720764, "rewards/judge_quality/std": 0.21224987506866455, "rewards/total_composite/mean": 0.8242409229278564, "rewards/total_composite/std": 0.0622052401304245, "reward": 0.8242409229278564, "reward_std": 0.0622052364051342, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09186495840549469, "sampling/sampling_logp_difference/max": 1.4436304569244385, "sampling/importance_sampling_ratio/min": 0.23606917262077332, "sampling/importance_sampling_ratio/mean": 1.001781702041626, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4357041157782078, "clip_ratio/low_mean": 0.07322037778794765, "clip_ratio/low_min": 0.07322037778794765, "clip_ratio/high_mean": 0.017804622184485197, "clip_ratio/high_max": 0.017804622184485197, "clip_ratio/region_mean": 0.09102499997243285, "reward_total_mean": 0.8242409229278564, "reward_meter_mean": 0.9816865921020508, "reward_meter_std": 0.014967069961130619, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9673194289207458, "reward_repeat_soft_std": 0.03114643134176731, "reward_judge_quality_mean": 0.45250001549720764, "reward_judge_quality_std": 0.21224987506866455, "reward_total_composite_mean": 0.8242409229278564, "reward_total_composite_std": 0.0622052401304245} {"timestamp_utc": "2026-04-13T04:50:38Z", "mode": "train", "global_step": 2579, "epoch": 0.2590657960823707, "loss": 0.041, "grad_norm": 14.530750274658203, "learning_rate": 2.187878787878788e-06, "num_tokens": 4792158.0, "completions/mean_length": 53.375, "completions/min_length": 49.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.375, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.8012256622314453, "rewards/meter/std": 0.32713282108306885, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9879710078239441, "rewards/repeat_soft/std": 0.009714987128973007, "rewards/judge_quality/mean": 0.42499998211860657, "rewards/judge_quality/std": 0.0707106739282608, "rewards/total_composite/mean": 0.7368485927581787, "rewards/total_composite/std": 0.15384620428085327, "reward": 0.7368485927581787, "reward_std": 0.15384620428085327, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1144632026553154, "sampling/sampling_logp_difference/max": 2.147101402282715, "sampling/importance_sampling_ratio/min": 0.11682228744029999, "sampling/importance_sampling_ratio/mean": 0.9911295771598816, "sampling/importance_sampling_ratio/max": 1.9250235557556152, "entropy": 0.49780984222888947, "clip_ratio/low_mean": 0.020979021675884724, "clip_ratio/low_min": 0.020979021675884724, "clip_ratio/high_mean": 0.09428057633340359, "clip_ratio/high_max": 0.09428057633340359, "clip_ratio/region_mean": 0.11525959800928831, "reward_total_mean": 0.7368485927581787, "reward_meter_mean": 0.8012256622314453, "reward_meter_std": 0.32713282108306885, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9879710078239441, "reward_repeat_soft_std": 0.009714987128973007, "reward_judge_quality_mean": 0.42499998211860657, "reward_judge_quality_std": 0.0707106739282608, "reward_total_composite_mean": 0.7368485927581787, "reward_total_composite_std": 0.15384620428085327} {"timestamp_utc": "2026-04-13T04:50:44Z", "mode": "train", "global_step": 2580, "epoch": 0.25916624811652433, "loss": 0.0258, "grad_norm": 8.080840110778809, "learning_rate": 2.184848484848485e-06, "num_tokens": 4793900.0, "completions/mean_length": 64.75, "completions/min_length": 60.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.75, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9734845161437988, "rewards/meter/std": 0.04642532765865326, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9571053981781006, "rewards/repeat_soft/std": 0.03543694689869881, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.8045285940170288, "rewards/total_composite/std": 0.024830905720591545, "reward": 0.8045285940170288, "reward_std": 0.0248308964073658, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10119166225194931, "sampling/sampling_logp_difference/max": 1.4056572914123535, "sampling/importance_sampling_ratio/min": 0.2452058345079422, "sampling/importance_sampling_ratio/mean": 0.9992998242378235, "sampling/importance_sampling_ratio/max": 1.8921723365783691, "entropy": 0.5180752761662006, "clip_ratio/low_mean": 0.030560662038624287, "clip_ratio/low_min": 0.030560662038624287, "clip_ratio/high_mean": 0.0880723474547267, "clip_ratio/high_max": 0.0880723474547267, "clip_ratio/region_mean": 0.11863300949335098, "reward_total_mean": 0.8045285940170288, "reward_meter_mean": 0.9734845161437988, "reward_meter_std": 0.04642532765865326, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9571053981781006, "reward_repeat_soft_std": 0.03543694689869881, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.8045285940170288, "reward_total_composite_std": 0.024830905720591545} {"timestamp_utc": "2026-04-13T04:50:55Z", "mode": "train", "global_step": 2581, "epoch": 0.25926670015067804, "loss": -0.2171, "grad_norm": 1.1418713331222534, "learning_rate": 2.181818181818182e-06, "num_tokens": 4796107.0, "completions/mean_length": 173.875, "completions/min_length": 114.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 125.5714340209961, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.9366760849952698, "rewards/meter/std": 0.15714594721794128, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8425620198249817, "rewards/repeat_soft/std": 0.10304437577724457, "rewards/judge_quality/mean": 0.19625000655651093, "rewards/judge_quality/std": 0.11070391535758972, "rewards/total_composite/mean": 0.6461269855499268, "rewards/total_composite/std": 0.2635821998119354, "reward": 0.6461269855499268, "reward_std": 0.2635821998119354, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06803137063980103, "sampling/sampling_logp_difference/max": 1.3079609870910645, "sampling/importance_sampling_ratio/min": 0.2703707814216614, "sampling/importance_sampling_ratio/mean": 0.9973156452178955, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.32315338775515556, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06933979410678148, "clip_ratio/high_max": 0.06933979410678148, "clip_ratio/region_mean": 0.06933979410678148, "reward_total_mean": 0.6461269855499268, "reward_meter_mean": 0.9366760849952698, "reward_meter_std": 0.15714594721794128, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8425620198249817, "reward_repeat_soft_std": 0.10304437577724457, "reward_judge_quality_mean": 0.19625000655651093, "reward_judge_quality_std": 0.11070391535758972, "reward_total_composite_mean": 0.6461269855499268, "reward_total_composite_std": 0.2635821998119354} {"timestamp_utc": "2026-04-13T04:51:05Z", "mode": "train", "global_step": 2582, "epoch": 0.25936715218483175, "loss": 0.0276, "grad_norm": 6.693411350250244, "learning_rate": 2.1787878787878788e-06, "num_tokens": 4798793.0, "completions/mean_length": 142.75, "completions/min_length": 134.0, "completions/max_length": 149.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.75, "completions/min_terminated_length": 134.0, "completions/max_terminated_length": 149.0, "rewards/meter/mean": 0.9877195358276367, "rewards/meter/std": 0.003539066994562745, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.793586015701294, "rewards/repeat_soft/std": 0.15036790072917938, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.7897074222564697, "rewards/total_composite/std": 0.041631586849689484, "reward": 0.7897074222564697, "reward_std": 0.04163156822323799, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08185619115829468, "sampling/sampling_logp_difference/max": 1.8648762702941895, "sampling/importance_sampling_ratio/min": 0.15491537749767303, "sampling/importance_sampling_ratio/mean": 1.007149338722229, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4427477829158306, "clip_ratio/low_mean": 0.0008445946150459349, "clip_ratio/low_min": 0.0008445946150459349, "clip_ratio/high_mean": 0.07762641459703445, "clip_ratio/high_max": 0.07762641459703445, "clip_ratio/region_mean": 0.07847100921208039, "reward_total_mean": 0.7897074222564697, "reward_meter_mean": 0.9877195358276367, "reward_meter_std": 0.003539066994562745, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.793586015701294, "reward_repeat_soft_std": 0.15036790072917938, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.7897074222564697, "reward_total_composite_std": 0.041631586849689484} {"timestamp_utc": "2026-04-13T04:51:11Z", "mode": "train", "global_step": 2583, "epoch": 0.25946760421898546, "loss": 0.0193, "grad_norm": 18.250057220458984, "learning_rate": 2.175757575757576e-06, "num_tokens": 4800247.0, "completions/mean_length": 29.75, "completions/min_length": 28.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.75, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.8932814598083496, "rewards/meter/std": 0.2684241235256195, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9536287188529968, "rewards/repeat_soft/std": 0.023023948073387146, "rewards/judge_quality/mean": 0.5600000023841858, "rewards/judge_quality/std": 0.22258226573467255, "rewards/total_composite/mean": 0.8153395652770996, "rewards/total_composite/std": 0.09236115217208862, "reward": 0.8153395652770996, "reward_std": 0.09236112982034683, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09089169651269913, "sampling/sampling_logp_difference/max": 1.6900922060012817, "sampling/importance_sampling_ratio/min": 0.184502512216568, "sampling/importance_sampling_ratio/mean": 1.0075650215148926, "sampling/importance_sampling_ratio/max": 1.5416053533554077, "entropy": 0.4964893236756325, "clip_ratio/low_mean": 0.01293103490024805, "clip_ratio/low_min": 0.01293103490024805, "clip_ratio/high_mean": 0.06231367541477084, "clip_ratio/high_max": 0.06231367541477084, "clip_ratio/region_mean": 0.07524471031501889, "reward_total_mean": 0.8153395652770996, "reward_meter_mean": 0.8932814598083496, "reward_meter_std": 0.2684241235256195, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9536287188529968, "reward_repeat_soft_std": 0.023023948073387146, "reward_judge_quality_mean": 0.5600000023841858, "reward_judge_quality_std": 0.22258226573467255, "reward_total_composite_mean": 0.8153395652770996, "reward_total_composite_std": 0.09236115217208862} {"timestamp_utc": "2026-04-13T04:51:18Z", "mode": "train", "global_step": 2584, "epoch": 0.2595680562531391, "loss": 0.0358, "grad_norm": 11.147346496582031, "learning_rate": 2.172727272727273e-06, "num_tokens": 4801968.0, "completions/mean_length": 60.125, "completions/min_length": 51.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.125, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.6621914505958557, "rewards/meter/std": 0.34742605686187744, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9695581197738647, "rewards/repeat_soft/std": 0.04059185832738876, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.6743170022964478, "rewards/total_composite/std": 0.15281172096729279, "reward": 0.6743170022964478, "reward_std": 0.1528117060661316, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12222409248352051, "sampling/sampling_logp_difference/max": 1.572899341583252, "sampling/importance_sampling_ratio/min": 0.20744286477565765, "sampling/importance_sampling_ratio/mean": 1.0002888441085815, "sampling/importance_sampling_ratio/max": 1.857171893119812, "entropy": 0.6992081627249718, "clip_ratio/low_mean": 0.06162083521485329, "clip_ratio/low_min": 0.06162083521485329, "clip_ratio/high_mean": 0.06571074016392231, "clip_ratio/high_max": 0.06571074016392231, "clip_ratio/region_mean": 0.1273315753787756, "reward_total_mean": 0.6743170022964478, "reward_meter_mean": 0.6621914505958557, "reward_meter_std": 0.34742605686187744, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9695581197738647, "reward_repeat_soft_std": 0.04059185832738876, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.6743170022964478, "reward_total_composite_std": 0.15281172096729279} {"timestamp_utc": "2026-04-13T04:51:24Z", "mode": "train", "global_step": 2585, "epoch": 0.2596685082872928, "loss": 0.0119, "grad_norm": 8.54994010925293, "learning_rate": 2.16969696969697e-06, "num_tokens": 4803706.0, "completions/mean_length": 57.25, "completions/min_length": 48.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.25, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9490251541137695, "rewards/meter/std": 0.10826536267995834, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9637717008590698, "rewards/repeat_soft/std": 0.045312028378248215, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.8579384088516235, "rewards/total_composite/std": 0.06946245580911636, "reward": 0.8579384088516235, "reward_std": 0.06946245580911636, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08833537995815277, "sampling/sampling_logp_difference/max": 1.4517147541046143, "sampling/importance_sampling_ratio/min": 0.2341683954000473, "sampling/importance_sampling_ratio/mean": 1.0059068202972412, "sampling/importance_sampling_ratio/max": 1.9107601642608643, "entropy": 0.49213187769055367, "clip_ratio/low_mean": 0.05182477971538901, "clip_ratio/low_min": 0.05182477971538901, "clip_ratio/high_mean": 0.02490096166729927, "clip_ratio/high_max": 0.02490096166729927, "clip_ratio/region_mean": 0.07672574138268828, "reward_total_mean": 0.8579384088516235, "reward_meter_mean": 0.9490251541137695, "reward_meter_std": 0.10826536267995834, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9637717008590698, "reward_repeat_soft_std": 0.045312028378248215, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.8579384088516235, "reward_total_composite_std": 0.06946245580911636} {"timestamp_utc": "2026-04-13T04:51:31Z", "mode": "train", "global_step": 2586, "epoch": 0.25976896032144653, "loss": 0.0316, "grad_norm": 6.994863510131836, "learning_rate": 2.166666666666667e-06, "num_tokens": 4805887.0, "completions/mean_length": 102.625, "completions/min_length": 93.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.625, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.9923177361488342, "rewards/meter/std": 0.0018325315322726965, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.95502769947052, "rewards/repeat_soft/std": 0.016940658912062645, "rewards/judge_quality/mean": 0.5575000047683716, "rewards/judge_quality/std": 0.19955310225486755, "rewards/total_composite/mean": 0.8592957258224487, "rewards/total_composite/std": 0.060391299426555634, "reward": 0.8592957258224487, "reward_std": 0.06039131432771683, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10312332212924957, "sampling/sampling_logp_difference/max": 3.1487491130828857, "sampling/importance_sampling_ratio/min": 0.042905762791633606, "sampling/importance_sampling_ratio/mean": 1.0025222301483154, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45847779512405396, "clip_ratio/low_mean": 0.049039970617741346, "clip_ratio/low_min": 0.049039970617741346, "clip_ratio/high_mean": 0.029354159953072667, "clip_ratio/high_max": 0.029354159953072667, "clip_ratio/region_mean": 0.07839413057081401, "reward_total_mean": 0.8592957258224487, "reward_meter_mean": 0.9923177361488342, "reward_meter_std": 0.0018325315322726965, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.95502769947052, "reward_repeat_soft_std": 0.016940658912062645, "reward_judge_quality_mean": 0.5575000047683716, "reward_judge_quality_std": 0.19955310225486755, "reward_total_composite_mean": 0.8592957258224487, "reward_total_composite_std": 0.060391299426555634} {"timestamp_utc": "2026-04-13T04:51:42Z", "mode": "train", "global_step": 2587, "epoch": 0.2598694123556002, "loss": -0.1311, "grad_norm": 1.6061720848083496, "learning_rate": 2.163636363636364e-06, "num_tokens": 4807475.0, "completions/mean_length": 171.5, "completions/min_length": 50.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.7829198241233826, "rewards/meter/std": 0.3831358850002289, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9793397188186646, "rewards/repeat_soft/std": 0.031144719570875168, "rewards/judge_quality/mean": 0.3425000011920929, "rewards/judge_quality/std": 0.18100906908512115, "rewards/total_composite/mean": 0.6184355020523071, "rewards/total_composite/std": 0.3817209005355835, "reward": 0.6184355020523071, "reward_std": 0.3817208707332611, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11312033236026764, "sampling/sampling_logp_difference/max": 1.726632833480835, "sampling/importance_sampling_ratio/min": 0.1778823584318161, "sampling/importance_sampling_ratio/mean": 1.0034291744232178, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4028352126479149, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09110348112881184, "clip_ratio/high_max": 0.09110348112881184, "clip_ratio/region_mean": 0.09110348112881184, "reward_total_mean": 0.6184355020523071, "reward_meter_mean": 0.7829198241233826, "reward_meter_std": 0.3831358850002289, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9793397188186646, "reward_repeat_soft_std": 0.031144719570875168, "reward_judge_quality_mean": 0.3425000011920929, "reward_judge_quality_std": 0.18100906908512115, "reward_total_composite_mean": 0.6184355020523071, "reward_total_composite_std": 0.3817209005355835} {"timestamp_utc": "2026-04-13T04:51:49Z", "mode": "train", "global_step": 2588, "epoch": 0.2599698643897539, "loss": 0.0118, "grad_norm": 6.705711841583252, "learning_rate": 2.1606060606060606e-06, "num_tokens": 4809727.0, "completions/mean_length": 99.5, "completions/min_length": 90.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.5, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.9625037312507629, "rewards/meter/std": 0.016227053478360176, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.675106942653656, "rewards/repeat_soft/std": 0.036896999925374985, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.759199857711792, "rewards/total_composite/std": 0.028467245399951935, "reward": 0.759199857711792, "reward_std": 0.028467250987887383, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04473431035876274, "sampling/sampling_logp_difference/max": 1.063964605331421, "sampling/importance_sampling_ratio/min": 0.3450849652290344, "sampling/importance_sampling_ratio/mean": 1.009466528892517, "sampling/importance_sampling_ratio/max": 1.9783036708831787, "entropy": 0.23619586788117886, "clip_ratio/low_mean": 0.021257080836221576, "clip_ratio/low_min": 0.021257080836221576, "clip_ratio/high_mean": 0.024939221097156405, "clip_ratio/high_max": 0.024939221097156405, "clip_ratio/region_mean": 0.04619630193337798, "reward_total_mean": 0.759199857711792, "reward_meter_mean": 0.9625037312507629, "reward_meter_std": 0.016227053478360176, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.675106942653656, "reward_repeat_soft_std": 0.036896999925374985, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.759199857711792, "reward_total_composite_std": 0.028467245399951935} {"timestamp_utc": "2026-04-13T04:52:00Z", "mode": "train", "global_step": 2589, "epoch": 0.2600703164239076, "loss": -0.0571, "grad_norm": 1.037161946296692, "learning_rate": 2.157575757575758e-06, "num_tokens": 4811254.0, "completions/mean_length": 208.875, "completions/min_length": 26.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 27.0, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.7294714450836182, "rewards/meter/std": 0.36677420139312744, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9443851709365845, "rewards/repeat_soft/std": 0.042360611259937286, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.43287867307662964, "rewards/total_composite/mean": 0.5816695690155029, "rewards/total_composite/std": 0.4844801425933838, "reward": 0.5816695690155029, "reward_std": 0.4844801127910614, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10425818711519241, "sampling/sampling_logp_difference/max": 2.2857229709625244, "sampling/importance_sampling_ratio/min": 0.10170051455497742, "sampling/importance_sampling_ratio/mean": 0.991552472114563, "sampling/importance_sampling_ratio/max": 1.45085871219635, "entropy": 0.2953361049294472, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.07445563841611147, "clip_ratio/high_max": 0.07445563841611147, "clip_ratio/region_mean": 0.07445563841611147, "reward_total_mean": 0.5816695690155029, "reward_meter_mean": 0.7294714450836182, "reward_meter_std": 0.36677420139312744, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9443851709365845, "reward_repeat_soft_std": 0.042360611259937286, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.43287867307662964, "reward_total_composite_mean": 0.5816695690155029, "reward_total_composite_std": 0.4844801425933838} {"timestamp_utc": "2026-04-13T04:52:11Z", "mode": "train", "global_step": 2590, "epoch": 0.26017076845806125, "loss": -0.1352, "grad_norm": 2.0361692905426025, "learning_rate": 2.1545454545454547e-06, "num_tokens": 4812808.0, "completions/mean_length": 172.25, "completions/min_length": 55.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 59.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9257738590240479, "rewards/meter/std": 0.10510864108800888, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.973411500453949, "rewards/repeat_soft/std": 0.026346616446971893, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.3390638530254364, "rewards/total_composite/mean": 0.6526162624359131, "rewards/total_composite/std": 0.40789228677749634, "reward": 0.6526162624359131, "reward_std": 0.40789228677749634, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09593217819929123, "sampling/sampling_logp_difference/max": 1.485269546508789, "sampling/importance_sampling_ratio/min": 0.22644129395484924, "sampling/importance_sampling_ratio/mean": 1.0284286737442017, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49371717125177383, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06631506746634841, "clip_ratio/high_max": 0.06631506746634841, "clip_ratio/region_mean": 0.06631506746634841, "reward_total_mean": 0.6526162624359131, "reward_meter_mean": 0.9257738590240479, "reward_meter_std": 0.10510864108800888, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.973411500453949, "reward_repeat_soft_std": 0.026346616446971893, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.3390638530254364, "reward_total_composite_mean": 0.6526162624359131, "reward_total_composite_std": 0.40789228677749634} {"timestamp_utc": "2026-04-13T04:52:18Z", "mode": "train", "global_step": 2591, "epoch": 0.26027122049221496, "loss": 0.0133, "grad_norm": 6.247508525848389, "learning_rate": 2.1515151515151515e-06, "num_tokens": 4814742.0, "completions/mean_length": 69.75, "completions/min_length": 65.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.75, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9889171123504639, "rewards/meter/std": 0.006624424830079079, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7303777933120728, "rewards/repeat_soft/std": 0.09353740513324738, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7940505146980286, "rewards/total_composite/std": 0.008869045414030552, "reward": 0.7940505146980286, "reward_std": 0.008869047276675701, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06598421931266785, "sampling/sampling_logp_difference/max": 1.908890724182129, "sampling/importance_sampling_ratio/min": 0.1482447385787964, "sampling/importance_sampling_ratio/mean": 1.0075353384017944, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3617117591202259, "clip_ratio/low_mean": 0.02327741403132677, "clip_ratio/low_min": 0.02327741403132677, "clip_ratio/high_mean": 0.03241564752534032, "clip_ratio/high_max": 0.03241564752534032, "clip_ratio/region_mean": 0.05569306155666709, "reward_total_mean": 0.7940505146980286, "reward_meter_mean": 0.9889171123504639, "reward_meter_std": 0.006624424830079079, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7303777933120728, "reward_repeat_soft_std": 0.09353740513324738, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7940505146980286, "reward_total_composite_std": 0.008869045414030552} {"timestamp_utc": "2026-04-13T04:52:31Z", "mode": "train", "global_step": 2592, "epoch": 0.26037167252636867, "loss": 0.0263, "grad_norm": 5.471983909606934, "learning_rate": 2.148484848484849e-06, "num_tokens": 4817338.0, "completions/mean_length": 135.5, "completions/min_length": 126.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 135.5, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.9664576053619385, "rewards/meter/std": 0.015847178176045418, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.745145320892334, "rewards/repeat_soft/std": 0.07065749168395996, "rewards/judge_quality/mean": 0.25874999165534973, "rewards/judge_quality/std": 0.07395702600479126, "rewards/total_composite/mean": 0.7070454359054565, "rewards/total_composite/std": 0.02426808327436447, "reward": 0.7070454359054565, "reward_std": 0.02426808513700962, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04911573976278305, "sampling/sampling_logp_difference/max": 1.7355341911315918, "sampling/importance_sampling_ratio/min": 0.17630599439144135, "sampling/importance_sampling_ratio/mean": 1.0057296752929688, "sampling/importance_sampling_ratio/max": 1.989497423171997, "entropy": 0.25779238156974316, "clip_ratio/low_mean": 0.037557785864919424, "clip_ratio/low_min": 0.037557785864919424, "clip_ratio/high_mean": 0.011118655558675528, "clip_ratio/high_max": 0.011118655558675528, "clip_ratio/region_mean": 0.04867644142359495, "reward_total_mean": 0.7070454359054565, "reward_meter_mean": 0.9664576053619385, "reward_meter_std": 0.015847178176045418, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.745145320892334, "reward_repeat_soft_std": 0.07065749168395996, "reward_judge_quality_mean": 0.25874999165534973, "reward_judge_quality_std": 0.07395702600479126, "reward_total_composite_mean": 0.7070454359054565, "reward_total_composite_std": 0.02426808327436447} {"timestamp_utc": "2026-04-13T04:52:38Z", "mode": "train", "global_step": 2593, "epoch": 0.2604721245605224, "loss": 0.0098, "grad_norm": 9.577892303466797, "learning_rate": 2.1454545454545456e-06, "num_tokens": 4819158.0, "completions/mean_length": 60.5, "completions/min_length": 54.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.5, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9847571849822998, "rewards/meter/std": 0.01916171796619892, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9852559566497803, "rewards/repeat_soft/std": 0.015238181687891483, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.8926663398742676, "rewards/total_composite/std": 0.08378516137599945, "reward": 0.8926663398742676, "reward_std": 0.08378516882658005, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09537127614021301, "sampling/sampling_logp_difference/max": 1.1614711284637451, "sampling/importance_sampling_ratio/min": 0.3130253553390503, "sampling/importance_sampling_ratio/mean": 1.0028724670410156, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4959523491561413, "clip_ratio/low_mean": 0.03157855407334864, "clip_ratio/low_min": 0.03157855407334864, "clip_ratio/high_mean": 0.0691692279651761, "clip_ratio/high_max": 0.0691692279651761, "clip_ratio/region_mean": 0.10074778203852475, "reward_total_mean": 0.8926663398742676, "reward_meter_mean": 0.9847571849822998, "reward_meter_std": 0.01916171796619892, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9852559566497803, "reward_repeat_soft_std": 0.015238181687891483, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.8926663398742676, "reward_total_composite_std": 0.08378516137599945} {"timestamp_utc": "2026-04-13T04:52:45Z", "mode": "train", "global_step": 2594, "epoch": 0.26057257659467603, "loss": -0.0157, "grad_norm": 23.281417846679688, "learning_rate": 2.1424242424242425e-06, "num_tokens": 4820631.0, "completions/mean_length": 19.125, "completions/min_length": 16.0, "completions/max_length": 22.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.125, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 22.0, "rewards/meter/mean": 0.9203278422355652, "rewards/meter/std": 0.02460252121090889, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.7822725176811218, "rewards/total_composite/std": 0.017125384882092476, "reward": 0.7822725176811218, "reward_std": 0.017125360667705536, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09350788593292236, "sampling/sampling_logp_difference/max": 1.175225019454956, "sampling/importance_sampling_ratio/min": 0.3087494969367981, "sampling/importance_sampling_ratio/mean": 1.0097748041152954, "sampling/importance_sampling_ratio/max": 1.6914223432540894, "entropy": 0.4283980317413807, "clip_ratio/low_mean": 0.06145334988832474, "clip_ratio/low_min": 0.06145334988832474, "clip_ratio/high_mean": 0.06520467950031161, "clip_ratio/high_max": 0.06520467950031161, "clip_ratio/region_mean": 0.12665802938863635, "reward_total_mean": 0.7822725176811218, "reward_meter_mean": 0.9203278422355652, "reward_meter_std": 0.02460252121090889, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.7822725176811218, "reward_total_composite_std": 0.017125384882092476} {"timestamp_utc": "2026-04-13T04:52:51Z", "mode": "train", "global_step": 2595, "epoch": 0.26067302862882974, "loss": 0.0043, "grad_norm": 12.296710014343262, "learning_rate": 2.1393939393939393e-06, "num_tokens": 4822303.0, "completions/mean_length": 45.0, "completions/min_length": 40.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.0, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.8826888799667358, "rewards/meter/std": 0.11662574112415314, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9871206283569336, "rewards/repeat_soft/std": 0.03375139832496643, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7741720676422119, "rewards/total_composite/std": 0.0536307729780674, "reward": 0.7741720676422119, "reward_std": 0.0536307729780674, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11740229278802872, "sampling/sampling_logp_difference/max": 1.6691689491271973, "sampling/importance_sampling_ratio/min": 0.18840357661247253, "sampling/importance_sampling_ratio/mean": 0.9927417635917664, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5403774678707123, "clip_ratio/low_mean": 0.02638888917863369, "clip_ratio/low_min": 0.02638888917863369, "clip_ratio/high_mean": 0.09513563197106123, "clip_ratio/high_max": 0.09513563197106123, "clip_ratio/region_mean": 0.12152452114969492, "reward_total_mean": 0.7741720676422119, "reward_meter_mean": 0.8826888799667358, "reward_meter_std": 0.11662574112415314, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9871206283569336, "reward_repeat_soft_std": 0.03375139832496643, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7741720676422119, "reward_total_composite_std": 0.0536307729780674} {"timestamp_utc": "2026-04-13T04:52:57Z", "mode": "train", "global_step": 2596, "epoch": 0.26077348066298345, "loss": 0.0204, "grad_norm": 10.455175399780273, "learning_rate": 2.1363636363636365e-06, "num_tokens": 4824086.0, "completions/mean_length": 58.875, "completions/min_length": 49.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.875, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9071290493011475, "rewards/meter/std": 0.14228737354278564, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9523197412490845, "rewards/repeat_soft/std": 0.015238041989505291, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.7741900682449341, "rewards/total_composite/std": 0.06500386446714401, "reward": 0.7741900682449341, "reward_std": 0.0650038793683052, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11145789176225662, "sampling/sampling_logp_difference/max": 1.0390496253967285, "sampling/importance_sampling_ratio/min": 0.3676456809043884, "sampling/importance_sampling_ratio/mean": 1.0131239891052246, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5082781575620174, "clip_ratio/low_mean": 0.05408300366252661, "clip_ratio/low_min": 0.05408300366252661, "clip_ratio/high_mean": 0.05993145424872637, "clip_ratio/high_max": 0.05993145424872637, "clip_ratio/region_mean": 0.11401445791125298, "reward_total_mean": 0.7741900682449341, "reward_meter_mean": 0.9071290493011475, "reward_meter_std": 0.14228737354278564, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9523197412490845, "reward_repeat_soft_std": 0.015238041989505291, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.7741900682449341, "reward_total_composite_std": 0.06500386446714401} {"timestamp_utc": "2026-04-13T04:53:05Z", "mode": "train", "global_step": 2597, "epoch": 0.2608739326971371, "loss": 0.0374, "grad_norm": 11.545970916748047, "learning_rate": 2.133333333333334e-06, "num_tokens": 4825591.0, "completions/mean_length": 42.125, "completions/min_length": 36.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.125, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.959794282913208, "rewards/meter/std": 0.011827902868390083, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9155590534210205, "rewards/repeat_soft/std": 0.10510583221912384, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8005883693695068, "rewards/total_composite/std": 0.008670798502862453, "reward": 0.8005883693695068, "reward_std": 0.008670811541378498, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0771254226565361, "sampling/sampling_logp_difference/max": 1.436112403869629, "sampling/importance_sampling_ratio/min": 0.2378506362438202, "sampling/importance_sampling_ratio/mean": 0.9981961250305176, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.384734321385622, "clip_ratio/low_mean": 0.02342282747849822, "clip_ratio/low_min": 0.02342282747849822, "clip_ratio/high_mean": 0.057152916211634874, "clip_ratio/high_max": 0.057152916211634874, "clip_ratio/region_mean": 0.0805757436901331, "reward_total_mean": 0.8005883693695068, "reward_meter_mean": 0.959794282913208, "reward_meter_std": 0.011827902868390083, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9155590534210205, "reward_repeat_soft_std": 0.10510583221912384, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8005883693695068, "reward_total_composite_std": 0.008670798502862453} {"timestamp_utc": "2026-04-13T04:53:12Z", "mode": "train", "global_step": 2598, "epoch": 0.2609743847312908, "loss": 0.0294, "grad_norm": 8.01186752319336, "learning_rate": 2.1303030303030306e-06, "num_tokens": 4827432.0, "completions/mean_length": 58.125, "completions/min_length": 52.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.125, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.985034704208374, "rewards/meter/std": 0.005730807315558195, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9341632723808289, "rewards/repeat_soft/std": 0.06217591091990471, "rewards/judge_quality/mean": 0.3999999761581421, "rewards/judge_quality/std": 0.09258200973272324, "rewards/total_composite/mean": 0.8066819906234741, "rewards/total_composite/std": 0.031292229890823364, "reward": 0.8066819906234741, "reward_std": 0.03129223734140396, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10120592266321182, "sampling/sampling_logp_difference/max": 1.469651222229004, "sampling/importance_sampling_ratio/min": 0.23000568151474, "sampling/importance_sampling_ratio/mean": 1.0214135646820068, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47718383371829987, "clip_ratio/low_mean": 0.02069672103971243, "clip_ratio/low_min": 0.02069672103971243, "clip_ratio/high_mean": 0.1020448487251997, "clip_ratio/high_max": 0.1020448487251997, "clip_ratio/region_mean": 0.12274156976491213, "reward_total_mean": 0.8066819906234741, "reward_meter_mean": 0.985034704208374, "reward_meter_std": 0.005730807315558195, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9341632723808289, "reward_repeat_soft_std": 0.06217591091990471, "reward_judge_quality_mean": 0.3999999761581421, "reward_judge_quality_std": 0.09258200973272324, "reward_total_composite_mean": 0.8066819906234741, "reward_total_composite_std": 0.031292229890823364} {"timestamp_utc": "2026-04-13T04:53:18Z", "mode": "train", "global_step": 2599, "epoch": 0.2610748367654445, "loss": -0.0338, "grad_norm": 17.18876838684082, "learning_rate": 2.1272727272727275e-06, "num_tokens": 4828773.0, "completions/mean_length": 21.625, "completions/min_length": 19.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.625, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.9015500545501709, "rewards/meter/std": 0.05263088271021843, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7846975326538086, "rewards/total_composite/std": 0.023973163217306137, "reward": 0.7846975326538086, "reward_std": 0.023973144590854645, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06814548373222351, "sampling/sampling_logp_difference/max": 0.9644970893859863, "sampling/importance_sampling_ratio/min": 0.38117486238479614, "sampling/importance_sampling_ratio/mean": 0.9984194040298462, "sampling/importance_sampling_ratio/max": 1.5619200468063354, "entropy": 0.27329748682677746, "clip_ratio/low_mean": 0.024463146924972534, "clip_ratio/low_min": 0.024463146924972534, "clip_ratio/high_mean": 0.050012941006571054, "clip_ratio/high_max": 0.050012941006571054, "clip_ratio/region_mean": 0.07447608793154359, "reward_total_mean": 0.7846975326538086, "reward_meter_mean": 0.9015500545501709, "reward_meter_std": 0.05263088271021843, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7846975326538086, "reward_total_composite_std": 0.023973163217306137} {"timestamp_utc": "2026-04-13T04:53:24Z", "mode": "train", "global_step": 2600, "epoch": 0.26117528879959817, "loss": 0.0229, "grad_norm": 10.155041694641113, "learning_rate": 2.1242424242424243e-06, "num_tokens": 4830356.0, "completions/mean_length": 51.875, "completions/min_length": 39.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.875, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.8717334270477295, "rewards/meter/std": 0.2602522373199463, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9014889001846313, "rewards/repeat_soft/std": 0.08302325755357742, "rewards/judge_quality/mean": 0.44749999046325684, "rewards/judge_quality/std": 0.12848013639450073, "rewards/total_composite/mean": 0.7666789293289185, "rewards/total_composite/std": 0.1472986787557602, "reward": 0.7666789293289185, "reward_std": 0.1472986787557602, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0963110700249672, "sampling/sampling_logp_difference/max": 1.3998959064483643, "sampling/importance_sampling_ratio/min": 0.24662263691425323, "sampling/importance_sampling_ratio/mean": 0.9900553822517395, "sampling/importance_sampling_ratio/max": 1.7982077598571777, "entropy": 0.3855416551232338, "clip_ratio/low_mean": 0.02495421376079321, "clip_ratio/low_min": 0.02495421376079321, "clip_ratio/high_mean": 0.07969336258247495, "clip_ratio/high_max": 0.07969336258247495, "clip_ratio/region_mean": 0.10464757634326816, "reward_total_mean": 0.7666789293289185, "reward_meter_mean": 0.8717334270477295, "reward_meter_std": 0.2602522373199463, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9014889001846313, "reward_repeat_soft_std": 0.08302325755357742, "reward_judge_quality_mean": 0.44749999046325684, "reward_judge_quality_std": 0.12848013639450073, "reward_total_composite_mean": 0.7666789293289185, "reward_total_composite_std": 0.1472986787557602} {"timestamp_utc": "2026-04-13T04:54:27Z", "mode": "eval", "global_step": 2600, "epoch": 0.26117528879959817, "eval_loss": NaN, "eval_runtime": 62.9731, "eval_samples_per_second": 1.27, "eval_steps_per_second": 0.159, "eval_num_tokens": 4830356.0, "eval_completions/mean_length": 111.6, "eval_completions/min_length": 45.8, "eval_completions/max_length": 251.0, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 100.8982147216797, "eval_completions/min_terminated_length": 45.8, "eval_completions/max_terminated_length": 172.6, "eval_rewards/meter/mean": 0.8763422667980194, "eval_rewards/meter/std": 0.21023509851656855, "eval_rewards/count_adherence/mean": 0.9718749940395355, "eval_rewards/count_adherence/std": 0.06978164613246918, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.07071067690849304, "eval_rewards/repeat_soft/mean": 0.870917671918869, "eval_rewards/repeat_soft/std": 0.12870224341750144, "eval_rewards/judge_quality/mean": 0.4183749884366989, "eval_rewards/judge_quality/std": 0.18403100445866585, "eval_rewards/total_composite/mean": 0.739558321237564, "eval_rewards/total_composite/std": 0.1488001558929682, "eval_reward": 0.739558321237564, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.03923238273710013, "eval_sampling/sampling_logp_difference/max": 0.8816334247589112, "eval_sampling/importance_sampling_ratio/min": 0.4210937529802322, "eval_sampling/importance_sampling_ratio/mean": 1.0061819434165955, "eval_sampling/importance_sampling_ratio/max": 1.3096721053123475, "eval_entropy": 0.3753200680017471, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.739558321237564, "eval_reward_meter_mean": 0.8763422667980194, "eval_reward_meter_std": 0.21023509851656855, "eval_reward_count_adherence_mean": 0.9718749940395355, "eval_reward_count_adherence_std": 0.06978164613246918, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.07071067690849304, "eval_reward_repeat_soft_mean": 0.870917671918869, "eval_reward_repeat_soft_std": 0.12870224341750144, "eval_reward_judge_quality_mean": 0.4183749884366989, "eval_reward_judge_quality_std": 0.18403100445866585, "eval_reward_total_composite_mean": 0.739558321237564, "eval_reward_total_composite_std": 0.1488001558929682} {"timestamp_utc": "2026-04-13T04:54:37Z", "mode": "train", "global_step": 2601, "epoch": 0.2612757408337519, "loss": 0.0421, "grad_norm": 10.008298873901367, "learning_rate": 2.1212121212121216e-06, "num_tokens": 4831920.0, "completions/mean_length": 41.5, "completions/min_length": 38.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.5, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9540509581565857, "rewards/meter/std": 0.016799384728074074, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8520238995552063, "rewards/repeat_soft/std": 0.11647067964076996, "rewards/judge_quality/mean": 0.35999998450279236, "rewards/judge_quality/std": 0.09165150672197342, "rewards/total_composite/mean": 0.7725253105163574, "rewards/total_composite/std": 0.032260291278362274, "reward": 0.7725253105163574, "reward_std": 0.03226028382778168, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06748487800359726, "sampling/sampling_logp_difference/max": 0.9806768894195557, "sampling/importance_sampling_ratio/min": 0.38049694895744324, "sampling/importance_sampling_ratio/mean": 1.0138072967529297, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3596693053841591, "clip_ratio/low_mean": 0.02655887114815414, "clip_ratio/low_min": 0.02655887114815414, "clip_ratio/high_mean": 0.046290030935779214, "clip_ratio/high_max": 0.046290030935779214, "clip_ratio/region_mean": 0.07284890208393335, "reward_total_mean": 0.7725253105163574, "reward_meter_mean": 0.9540509581565857, "reward_meter_std": 0.016799384728074074, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8520238995552063, "reward_repeat_soft_std": 0.11647067964076996, "reward_judge_quality_mean": 0.35999998450279236, "reward_judge_quality_std": 0.09165150672197342, "reward_total_composite_mean": 0.7725253105163574, "reward_total_composite_std": 0.032260291278362274} {"timestamp_utc": "2026-04-13T04:54:43Z", "mode": "train", "global_step": 2602, "epoch": 0.2613761928679056, "loss": -0.0027, "grad_norm": 11.016915321350098, "learning_rate": 2.1181818181818184e-06, "num_tokens": 4833319.0, "completions/mean_length": 25.875, "completions/min_length": 24.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.875, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.8465481400489807, "rewards/meter/std": 0.08326733112335205, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7396677136421204, "rewards/repeat_soft/std": 0.0970526710152626, "rewards/judge_quality/mean": 0.6812499761581421, "rewards/judge_quality/std": 0.25542333722114563, "rewards/total_composite/mean": 0.8092883825302124, "rewards/total_composite/std": 0.08725832402706146, "reward": 0.8092883825302124, "reward_std": 0.08725833147764206, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05200088396668434, "sampling/sampling_logp_difference/max": 1.2117526531219482, "sampling/importance_sampling_ratio/min": 0.29767510294914246, "sampling/importance_sampling_ratio/mean": 1.002990961074829, "sampling/importance_sampling_ratio/max": 1.762626051902771, "entropy": 0.20263521373271942, "clip_ratio/low_mean": 0.024038461968302727, "clip_ratio/low_min": 0.024038461968302727, "clip_ratio/high_mean": 0.039930555038154125, "clip_ratio/high_max": 0.039930555038154125, "clip_ratio/region_mean": 0.06396901700645685, "reward_total_mean": 0.8092883825302124, "reward_meter_mean": 0.8465481400489807, "reward_meter_std": 0.08326733112335205, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7396677136421204, "reward_repeat_soft_std": 0.0970526710152626, "reward_judge_quality_mean": 0.6812499761581421, "reward_judge_quality_std": 0.25542333722114563, "reward_total_composite_mean": 0.8092883825302124, "reward_total_composite_std": 0.08725832402706146} {"timestamp_utc": "2026-04-13T04:54:49Z", "mode": "train", "global_step": 2603, "epoch": 0.26147664490205924, "loss": 0.1014, "grad_norm": 16.9239444732666, "learning_rate": 2.1151515151515152e-06, "num_tokens": 4834950.0, "completions/mean_length": 32.875, "completions/min_length": 26.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.875, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.6756708025932312, "rewards/meter/std": 0.3539947271347046, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9358371496200562, "rewards/repeat_soft/std": 0.03892217203974724, "rewards/judge_quality/mean": 0.3675000071525574, "rewards/judge_quality/std": 0.1348808854818344, "rewards/total_composite/mean": 0.6578855514526367, "rewards/total_composite/std": 0.15195775032043457, "reward": 0.6578855514526367, "reward_std": 0.15195773541927338, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1565026044845581, "sampling/sampling_logp_difference/max": 2.441702365875244, "sampling/importance_sampling_ratio/min": 0.08701258897781372, "sampling/importance_sampling_ratio/mean": 1.006704568862915, "sampling/importance_sampling_ratio/max": 1.8712174892425537, "entropy": 0.9447309896349907, "clip_ratio/low_mean": 0.04529138468205929, "clip_ratio/low_min": 0.04529138468205929, "clip_ratio/high_mean": 0.09242453146725893, "clip_ratio/high_max": 0.09242453146725893, "clip_ratio/region_mean": 0.13771591614931822, "reward_total_mean": 0.6578855514526367, "reward_meter_mean": 0.6756708025932312, "reward_meter_std": 0.3539947271347046, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9358371496200562, "reward_repeat_soft_std": 0.03892217203974724, "reward_judge_quality_mean": 0.3675000071525574, "reward_judge_quality_std": 0.1348808854818344, "reward_total_composite_mean": 0.6578855514526367, "reward_total_composite_std": 0.15195775032043457} {"timestamp_utc": "2026-04-13T04:55:00Z", "mode": "train", "global_step": 2604, "epoch": 0.26157709693621295, "loss": -0.0023, "grad_norm": 3.3952620029449463, "learning_rate": 2.1121212121212125e-06, "num_tokens": 4837859.0, "completions/mean_length": 180.625, "completions/min_length": 160.0, "completions/max_length": 191.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 180.625, "completions/min_terminated_length": 160.0, "completions/max_terminated_length": 191.0, "rewards/meter/mean": 0.9942585229873657, "rewards/meter/std": 0.0007477445178665221, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7943666577339172, "rewards/repeat_soft/std": 0.1320447325706482, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.790103018283844, "rewards/total_composite/std": 0.035705745220184326, "reward": 0.790103018283844, "reward_std": 0.035705745220184326, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06333569437265396, "sampling/sampling_logp_difference/max": 1.5444830656051636, "sampling/importance_sampling_ratio/min": 0.21342216432094574, "sampling/importance_sampling_ratio/mean": 1.0086349248886108, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3192886523902416, "clip_ratio/low_mean": 0.010613359045237303, "clip_ratio/low_min": 0.010613359045237303, "clip_ratio/high_mean": 0.04514891607686877, "clip_ratio/high_max": 0.04514891607686877, "clip_ratio/region_mean": 0.055762275122106075, "reward_total_mean": 0.790103018283844, "reward_meter_mean": 0.9942585229873657, "reward_meter_std": 0.0007477445178665221, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7943666577339172, "reward_repeat_soft_std": 0.1320447325706482, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.790103018283844, "reward_total_composite_std": 0.035705745220184326} {"timestamp_utc": "2026-04-13T04:55:06Z", "mode": "train", "global_step": 2605, "epoch": 0.26167754897036666, "loss": 0.0225, "grad_norm": 5.012233734130859, "learning_rate": 2.1090909090909093e-06, "num_tokens": 4839502.0, "completions/mean_length": 57.375, "completions/min_length": 53.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.375, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.8957502841949463, "rewards/meter/std": 0.22335083782672882, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.940125584602356, "rewards/repeat_soft/std": 0.033887963742017746, "rewards/judge_quality/mean": 0.4312499761581421, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.776475191116333, "rewards/total_composite/std": 0.10128656029701233, "reward": 0.776475191116333, "reward_std": 0.10128655284643173, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09811055660247803, "sampling/sampling_logp_difference/max": 2.216866970062256, "sampling/importance_sampling_ratio/min": 0.10894991457462311, "sampling/importance_sampling_ratio/mean": 0.9974266886711121, "sampling/importance_sampling_ratio/max": 1.9591768980026245, "entropy": 0.384040292352438, "clip_ratio/low_mean": 0.010593220591545105, "clip_ratio/low_min": 0.010593220591545105, "clip_ratio/high_mean": 0.06331329699605703, "clip_ratio/high_max": 0.06331329699605703, "clip_ratio/region_mean": 0.07390651758760214, "reward_total_mean": 0.776475191116333, "reward_meter_mean": 0.8957502841949463, "reward_meter_std": 0.22335083782672882, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.940125584602356, "reward_repeat_soft_std": 0.033887963742017746, "reward_judge_quality_mean": 0.4312499761581421, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.776475191116333, "reward_total_composite_std": 0.10128656029701233} {"timestamp_utc": "2026-04-13T04:55:18Z", "mode": "train", "global_step": 2606, "epoch": 0.26177800100452037, "loss": -0.059, "grad_norm": 1.0530824661254883, "learning_rate": 2.106060606060606e-06, "num_tokens": 4840938.0, "completions/mean_length": 151.5, "completions/min_length": 28.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 31.33333396911621, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.933724045753479, "rewards/meter/std": 0.11239439994096756, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9617863893508911, "rewards/repeat_soft/std": 0.0014362066285684705, "rewards/judge_quality/mean": 0.3387500047683716, "rewards/judge_quality/std": 0.1787606030702591, "rewards/total_composite/mean": 0.6849676370620728, "rewards/total_composite/std": 0.28327956795692444, "reward": 0.6849676370620728, "reward_std": 0.2832795977592468, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11282263696193695, "sampling/sampling_logp_difference/max": 1.6075270175933838, "sampling/importance_sampling_ratio/min": 0.2003825455904007, "sampling/importance_sampling_ratio/mean": 1.0008995532989502, "sampling/importance_sampling_ratio/max": 1.9077956676483154, "entropy": 0.3703305013477802, "clip_ratio/low_mean": 0.0078125, "clip_ratio/low_min": 0.0078125, "clip_ratio/high_mean": 0.08251925930380821, "clip_ratio/high_max": 0.08251925930380821, "clip_ratio/region_mean": 0.09033175930380821, "reward_total_mean": 0.6849676370620728, "reward_meter_mean": 0.933724045753479, "reward_meter_std": 0.11239439994096756, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9617863893508911, "reward_repeat_soft_std": 0.0014362066285684705, "reward_judge_quality_mean": 0.3387500047683716, "reward_judge_quality_std": 0.1787606030702591, "reward_total_composite_mean": 0.6849676370620728, "reward_total_composite_std": 0.28327956795692444} {"timestamp_utc": "2026-04-13T04:55:25Z", "mode": "train", "global_step": 2607, "epoch": 0.261878453038674, "loss": 0.0549, "grad_norm": 6.65385103225708, "learning_rate": 2.103030303030303e-06, "num_tokens": 4843320.0, "completions/mean_length": 108.75, "completions/min_length": 87.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.75, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.7216966152191162, "rewards/meter/std": 0.2006174474954605, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8416332006454468, "rewards/repeat_soft/std": 0.05264674127101898, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.21357084810733795, "rewards/total_composite/mean": 0.6823017597198486, "rewards/total_composite/std": 0.11838166415691376, "reward": 0.6823017597198486, "reward_std": 0.11838166415691376, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09222668409347534, "sampling/sampling_logp_difference/max": 1.8639137744903564, "sampling/importance_sampling_ratio/min": 0.1831294149160385, "sampling/importance_sampling_ratio/mean": 0.9989333748817444, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3460864834487438, "clip_ratio/low_mean": 0.03877254901453853, "clip_ratio/low_min": 0.03877254901453853, "clip_ratio/high_mean": 0.04662430612370372, "clip_ratio/high_max": 0.04662430612370372, "clip_ratio/region_mean": 0.08539685513824224, "reward_total_mean": 0.6823017597198486, "reward_meter_mean": 0.7216966152191162, "reward_meter_std": 0.2006174474954605, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8416332006454468, "reward_repeat_soft_std": 0.05264674127101898, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.21357084810733795, "reward_total_composite_mean": 0.6823017597198486, "reward_total_composite_std": 0.11838166415691376} {"timestamp_utc": "2026-04-13T04:55:31Z", "mode": "train", "global_step": 2608, "epoch": 0.26197890507282773, "loss": 0.0728, "grad_norm": 18.05392837524414, "learning_rate": 2.1000000000000002e-06, "num_tokens": 4844600.0, "completions/mean_length": 25.0, "completions/min_length": 21.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.0, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.8258306980133057, "rewards/meter/std": 0.34174543619155884, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9346997141838074, "rewards/repeat_soft/std": 0.0589936226606369, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.7632187604904175, "rewards/total_composite/std": 0.1692786067724228, "reward": 0.7632187604904175, "reward_std": 0.1692786067724228, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09709241986274719, "sampling/sampling_logp_difference/max": 1.304032802581787, "sampling/importance_sampling_ratio/min": 0.2714349627494812, "sampling/importance_sampling_ratio/mean": 0.99810791015625, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3627129979431629, "clip_ratio/low_mean": 0.02281746082007885, "clip_ratio/low_min": 0.02281746082007885, "clip_ratio/high_mean": 0.07805781345814466, "clip_ratio/high_max": 0.07805781345814466, "clip_ratio/region_mean": 0.10087527427822351, "reward_total_mean": 0.7632187604904175, "reward_meter_mean": 0.8258306980133057, "reward_meter_std": 0.34174543619155884, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9346997141838074, "reward_repeat_soft_std": 0.0589936226606369, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.7632187604904175, "reward_total_composite_std": 0.1692786067724228} {"timestamp_utc": "2026-04-13T04:55:38Z", "mode": "train", "global_step": 2609, "epoch": 0.26207935710698144, "loss": 0.013, "grad_norm": 15.775561332702637, "learning_rate": 2.096969696969697e-06, "num_tokens": 4846177.0, "completions/mean_length": 52.125, "completions/min_length": 45.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.125, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.8206400871276855, "rewards/meter/std": 0.20741736888885498, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9280018210411072, "rewards/repeat_soft/std": 0.05500160902738571, "rewards/judge_quality/mean": 0.38874998688697815, "rewards/judge_quality/std": 0.08675704896450043, "rewards/total_composite/mean": 0.7287132143974304, "rewards/total_composite/std": 0.08557647466659546, "reward": 0.7287132143974304, "reward_std": 0.08557647466659546, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09915075451135635, "sampling/sampling_logp_difference/max": 2.2943618297576904, "sampling/importance_sampling_ratio/min": 0.1008257195353508, "sampling/importance_sampling_ratio/mean": 1.0001474618911743, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4682883284986019, "clip_ratio/low_mean": 0.035342082381248474, "clip_ratio/low_min": 0.035342082381248474, "clip_ratio/high_mean": 0.06326810200698674, "clip_ratio/high_max": 0.06326810200698674, "clip_ratio/region_mean": 0.09861018438823521, "reward_total_mean": 0.7287132143974304, "reward_meter_mean": 0.8206400871276855, "reward_meter_std": 0.20741736888885498, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9280018210411072, "reward_repeat_soft_std": 0.05500160902738571, "reward_judge_quality_mean": 0.38874998688697815, "reward_judge_quality_std": 0.08675704896450043, "reward_total_composite_mean": 0.7287132143974304, "reward_total_composite_std": 0.08557647466659546} {"timestamp_utc": "2026-04-13T04:55:44Z", "mode": "train", "global_step": 2610, "epoch": 0.2621798091411351, "loss": -0.023, "grad_norm": 14.493708610534668, "learning_rate": 2.093939393939394e-06, "num_tokens": 4847839.0, "completions/mean_length": 47.75, "completions/min_length": 41.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.75, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.8613373041152954, "rewards/meter/std": 0.21981169283390045, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8449946641921997, "rewards/repeat_soft/std": 0.13042543828487396, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7548512816429138, "rewards/total_composite/std": 0.09594916552305222, "reward": 0.7548512816429138, "reward_std": 0.09594918042421341, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10176989436149597, "sampling/sampling_logp_difference/max": 2.59786057472229, "sampling/importance_sampling_ratio/min": 0.07443264871835709, "sampling/importance_sampling_ratio/mean": 0.9973095059394836, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.46859167143702507, "clip_ratio/low_mean": 0.03245257493108511, "clip_ratio/low_min": 0.03245257493108511, "clip_ratio/high_mean": 0.08148965612053871, "clip_ratio/high_max": 0.08148965612053871, "clip_ratio/region_mean": 0.11394223105162382, "reward_total_mean": 0.7548512816429138, "reward_meter_mean": 0.8613373041152954, "reward_meter_std": 0.21981169283390045, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8449946641921997, "reward_repeat_soft_std": 0.13042543828487396, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7548512816429138, "reward_total_composite_std": 0.09594916552305222} {"timestamp_utc": "2026-04-13T04:55:51Z", "mode": "train", "global_step": 2611, "epoch": 0.2622802611752888, "loss": 0.1799, "grad_norm": 10.316991806030273, "learning_rate": 2.090909090909091e-06, "num_tokens": 4849606.0, "completions/mean_length": 61.875, "completions/min_length": 53.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.875, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.958871603012085, "rewards/meter/std": 0.08095888793468475, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9672017097473145, "rewards/repeat_soft/std": 0.016964295879006386, "rewards/judge_quality/mean": 0.3974999785423279, "rewards/judge_quality/std": 0.11310551315546036, "rewards/total_composite/mean": 0.7974624037742615, "rewards/total_composite/std": 0.07097756862640381, "reward": 0.7974624037742615, "reward_std": 0.07097756862640381, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08370936661958694, "sampling/sampling_logp_difference/max": 1.4863775968551636, "sampling/importance_sampling_ratio/min": 0.22619052231311798, "sampling/importance_sampling_ratio/mean": 1.0021541118621826, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8414957709610462, "clip_ratio/low_mean": 0.005376344081014395, "clip_ratio/low_min": 0.005376344081014395, "clip_ratio/high_mean": 0.07540695741772652, "clip_ratio/high_max": 0.07540695741772652, "clip_ratio/region_mean": 0.08078330149874091, "reward_total_mean": 0.7974624037742615, "reward_meter_mean": 0.958871603012085, "reward_meter_std": 0.08095888793468475, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9672017097473145, "reward_repeat_soft_std": 0.016964295879006386, "reward_judge_quality_mean": 0.3974999785423279, "reward_judge_quality_std": 0.11310551315546036, "reward_total_composite_mean": 0.7974624037742615, "reward_total_composite_std": 0.07097756862640381} {"timestamp_utc": "2026-04-13T04:55:57Z", "mode": "train", "global_step": 2612, "epoch": 0.2623807132094425, "loss": 0.0289, "grad_norm": 13.392422676086426, "learning_rate": 2.087878787878788e-06, "num_tokens": 4851325.0, "completions/mean_length": 48.875, "completions/min_length": 44.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.875, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9599260091781616, "rewards/meter/std": 0.02947339229285717, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8682520985603333, "rewards/repeat_soft/std": 0.11990659683942795, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.17353467643260956, "rewards/total_composite/mean": 0.7917919158935547, "rewards/total_composite/std": 0.05234923213720322, "reward": 0.7917919158935547, "reward_std": 0.05234923213720322, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08794523030519485, "sampling/sampling_logp_difference/max": 1.6394377946853638, "sampling/importance_sampling_ratio/min": 0.1940891295671463, "sampling/importance_sampling_ratio/mean": 1.0057512521743774, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.38760681450366974, "clip_ratio/low_mean": 0.03082204656675458, "clip_ratio/low_min": 0.03082204656675458, "clip_ratio/high_mean": 0.059817928122356534, "clip_ratio/high_max": 0.059817928122356534, "clip_ratio/region_mean": 0.09063997468911111, "reward_total_mean": 0.7917919158935547, "reward_meter_mean": 0.9599260091781616, "reward_meter_std": 0.02947339229285717, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8682520985603333, "reward_repeat_soft_std": 0.11990659683942795, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.17353467643260956, "reward_total_composite_mean": 0.7917919158935547, "reward_total_composite_std": 0.05234923213720322} {"timestamp_utc": "2026-04-13T04:56:04Z", "mode": "train", "global_step": 2613, "epoch": 0.26248116524359616, "loss": 0.0193, "grad_norm": 5.199469566345215, "learning_rate": 2.0848484848484852e-06, "num_tokens": 4853716.0, "completions/mean_length": 121.875, "completions/min_length": 109.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 121.875, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.9919331073760986, "rewards/meter/std": 0.0026154161896556616, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8722885251045227, "rewards/repeat_soft/std": 0.043912459164857864, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.803223729133606, "rewards/total_composite/std": 0.01888357102870941, "reward": 0.803223729133606, "reward_std": 0.018883569166064262, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06777837127447128, "sampling/sampling_logp_difference/max": 2.119602680206299, "sampling/importance_sampling_ratio/min": 0.12007932364940643, "sampling/importance_sampling_ratio/mean": 1.0166826248168945, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.30554838851094246, "clip_ratio/low_mean": 0.021319672465324402, "clip_ratio/low_min": 0.021319672465324402, "clip_ratio/high_mean": 0.04422337794676423, "clip_ratio/high_max": 0.04422337794676423, "clip_ratio/region_mean": 0.06554305041208863, "reward_total_mean": 0.803223729133606, "reward_meter_mean": 0.9919331073760986, "reward_meter_std": 0.0026154161896556616, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8722885251045227, "reward_repeat_soft_std": 0.043912459164857864, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.803223729133606, "reward_total_composite_std": 0.01888357102870941} {"timestamp_utc": "2026-04-13T04:56:11Z", "mode": "train", "global_step": 2614, "epoch": 0.26258161727774987, "loss": -0.0038, "grad_norm": 5.053582191467285, "learning_rate": 2.081818181818182e-06, "num_tokens": 4856158.0, "completions/mean_length": 128.25, "completions/min_length": 124.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.25, "completions/min_terminated_length": 124.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.990971565246582, "rewards/meter/std": 0.0018077600980177522, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8894104361534119, "rewards/repeat_soft/std": 0.04685764014720917, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.7816282510757446, "rewards/total_composite/std": 0.03401906043291092, "reward": 0.7816282510757446, "reward_std": 0.034019071608781815, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06686059385538101, "sampling/sampling_logp_difference/max": 2.0626380443573, "sampling/importance_sampling_ratio/min": 0.12711818516254425, "sampling/importance_sampling_ratio/mean": 1.0088887214660645, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37041769176721573, "clip_ratio/low_mean": 0.029852701351046562, "clip_ratio/low_min": 0.029852701351046562, "clip_ratio/high_mean": 0.04224513843655586, "clip_ratio/high_max": 0.04224513843655586, "clip_ratio/region_mean": 0.07209783978760242, "reward_total_mean": 0.7816282510757446, "reward_meter_mean": 0.990971565246582, "reward_meter_std": 0.0018077600980177522, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8894104361534119, "reward_repeat_soft_std": 0.04685764014720917, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.7816282510757446, "reward_total_composite_std": 0.03401906043291092} {"timestamp_utc": "2026-04-13T04:56:17Z", "mode": "train", "global_step": 2615, "epoch": 0.2626820693119036, "loss": 0.0642, "grad_norm": 15.350176811218262, "learning_rate": 2.078787878787879e-06, "num_tokens": 4857690.0, "completions/mean_length": 27.5, "completions/min_length": 23.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.5, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9850263595581055, "rewards/meter/std": 0.0031075740698724985, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9311496019363403, "rewards/repeat_soft/std": 0.05829960107803345, "rewards/judge_quality/mean": 0.44999998807907104, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8213768005371094, "rewards/total_composite/std": 0.006431314628571272, "reward": 0.8213768005371094, "reward_std": 0.006431326735764742, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08061274886131287, "sampling/sampling_logp_difference/max": 1.2717137336730957, "sampling/importance_sampling_ratio/min": 0.28035077452659607, "sampling/importance_sampling_ratio/mean": 0.9981029629707336, "sampling/importance_sampling_ratio/max": 1.460128664970398, "entropy": 0.43031924590468407, "clip_ratio/low_mean": 0.02578671369701624, "clip_ratio/low_min": 0.02578671369701624, "clip_ratio/high_mean": 0.09135789051651955, "clip_ratio/high_max": 0.09135789051651955, "clip_ratio/region_mean": 0.11714460421353579, "reward_total_mean": 0.8213768005371094, "reward_meter_mean": 0.9850263595581055, "reward_meter_std": 0.0031075740698724985, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9311496019363403, "reward_repeat_soft_std": 0.05829960107803345, "reward_judge_quality_mean": 0.44999998807907104, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8213768005371094, "reward_total_composite_std": 0.006431314628571272} {"timestamp_utc": "2026-04-13T04:56:24Z", "mode": "train", "global_step": 2616, "epoch": 0.2627825213460573, "loss": 0.0318, "grad_norm": 5.645806312561035, "learning_rate": 2.075757575757576e-06, "num_tokens": 4860049.0, "completions/mean_length": 117.875, "completions/min_length": 114.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.875, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.9860020875930786, "rewards/meter/std": 0.004484372679144144, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8998234272003174, "rewards/repeat_soft/std": 0.07337522506713867, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.8584332466125488, "rewards/total_composite/std": 0.07293303310871124, "reward": 0.8584332466125488, "reward_std": 0.07293303310871124, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08611883223056793, "sampling/sampling_logp_difference/max": 4.433371543884277, "sampling/importance_sampling_ratio/min": 0.011874387040734291, "sampling/importance_sampling_ratio/mean": 1.0007350444793701, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3712894693017006, "clip_ratio/low_mean": 0.05131254391744733, "clip_ratio/low_min": 0.05131254391744733, "clip_ratio/high_mean": 0.038920831866562366, "clip_ratio/high_max": 0.038920831866562366, "clip_ratio/region_mean": 0.0902333757840097, "reward_total_mean": 0.8584332466125488, "reward_meter_mean": 0.9860020875930786, "reward_meter_std": 0.004484372679144144, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8998234272003174, "reward_repeat_soft_std": 0.07337522506713867, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.8584332466125488, "reward_total_composite_std": 0.07293303310871124} {"timestamp_utc": "2026-04-13T04:56:30Z", "mode": "train", "global_step": 2617, "epoch": 0.26288297338021094, "loss": 0.1326, "grad_norm": 25.410263061523438, "learning_rate": 2.072727272727273e-06, "num_tokens": 4861450.0, "completions/mean_length": 22.125, "completions/min_length": 19.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.125, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9240397214889526, "rewards/meter/std": 0.02638438157737255, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9517762660980225, "rewards/repeat_soft/std": 0.02931642159819603, "rewards/judge_quality/mean": 0.3462499976158142, "rewards/judge_quality/std": 0.10336308926343918, "rewards/total_composite/mean": 0.7461205124855042, "rewards/total_composite/std": 0.04549362510442734, "reward": 0.7461205124855042, "reward_std": 0.04549361392855644, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10024179518222809, "sampling/sampling_logp_difference/max": 1.2783472537994385, "sampling/importance_sampling_ratio/min": 0.27849718928337097, "sampling/importance_sampling_ratio/mean": 1.004619836807251, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.461125485599041, "clip_ratio/low_mean": 0.06136363744735718, "clip_ratio/low_min": 0.06136363744735718, "clip_ratio/high_mean": 0.023918492253869772, "clip_ratio/high_max": 0.023918492253869772, "clip_ratio/region_mean": 0.08528212970122695, "reward_total_mean": 0.7461205124855042, "reward_meter_mean": 0.9240397214889526, "reward_meter_std": 0.02638438157737255, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9517762660980225, "reward_repeat_soft_std": 0.02931642159819603, "reward_judge_quality_mean": 0.3462499976158142, "reward_judge_quality_std": 0.10336308926343918, "reward_total_composite_mean": 0.7461205124855042, "reward_total_composite_std": 0.04549362510442734} {"timestamp_utc": "2026-04-13T04:56:37Z", "mode": "train", "global_step": 2618, "epoch": 0.26298342541436465, "loss": 0.0278, "grad_norm": 6.1044840812683105, "learning_rate": 2.06969696969697e-06, "num_tokens": 4863871.0, "completions/mean_length": 124.625, "completions/min_length": 116.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 124.625, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.9037842750549316, "rewards/meter/std": 0.22064973413944244, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9127162098884583, "rewards/repeat_soft/std": 0.03997720032930374, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.7484745383262634, "rewards/total_composite/std": 0.09187249094247818, "reward": 0.7484745383262634, "reward_std": 0.09187250584363937, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07252392172813416, "sampling/sampling_logp_difference/max": 1.3273894786834717, "sampling/importance_sampling_ratio/min": 0.26516857743263245, "sampling/importance_sampling_ratio/mean": 1.0022121667861938, "sampling/importance_sampling_ratio/max": 1.8704707622528076, "entropy": 0.3935665003955364, "clip_ratio/low_mean": 0.013811899349093437, "clip_ratio/low_min": 0.013811899349093437, "clip_ratio/high_mean": 0.046538058668375015, "clip_ratio/high_max": 0.046538058668375015, "clip_ratio/region_mean": 0.06034995801746845, "reward_total_mean": 0.7484745383262634, "reward_meter_mean": 0.9037842750549316, "reward_meter_std": 0.22064973413944244, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9127162098884583, "reward_repeat_soft_std": 0.03997720032930374, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.7484745383262634, "reward_total_composite_std": 0.09187249094247818} {"timestamp_utc": "2026-04-13T04:56:44Z", "mode": "train", "global_step": 2619, "epoch": 0.26308387744851836, "loss": 0.0357, "grad_norm": 7.5360870361328125, "learning_rate": 2.0666666666666666e-06, "num_tokens": 4866291.0, "completions/mean_length": 105.5, "completions/min_length": 95.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.5, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.9028578400611877, "rewards/meter/std": 0.14707715809345245, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9317258596420288, "rewards/repeat_soft/std": 0.03159402683377266, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.7878336310386658, "rewards/total_composite/std": 0.08790010213851929, "reward": 0.7878336310386658, "reward_std": 0.0879000872373581, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08656974136829376, "sampling/sampling_logp_difference/max": 1.8268406391143799, "sampling/importance_sampling_ratio/min": 0.2695920467376709, "sampling/importance_sampling_ratio/mean": 1.0021508932113647, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37096916511654854, "clip_ratio/low_mean": 0.01802752260118723, "clip_ratio/low_min": 0.01802752260118723, "clip_ratio/high_mean": 0.06970652285963297, "clip_ratio/high_max": 0.06970652285963297, "clip_ratio/region_mean": 0.0877340454608202, "reward_total_mean": 0.7878336310386658, "reward_meter_mean": 0.9028578400611877, "reward_meter_std": 0.14707715809345245, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9317258596420288, "reward_repeat_soft_std": 0.03159402683377266, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.7878336310386658, "reward_total_composite_std": 0.08790010213851929} {"timestamp_utc": "2026-04-13T04:56:56Z", "mode": "train", "global_step": 2620, "epoch": 0.263184329482672, "loss": -0.0873, "grad_norm": 1.8680567741394043, "learning_rate": 2.063636363636364e-06, "num_tokens": 4867631.0, "completions/mean_length": 86.5, "completions/min_length": 23.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 25.71428680419922, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.8535219430923462, "rewards/meter/std": 0.3448048233985901, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9382025003433228, "rewards/repeat_soft/std": 0.06872370839118958, "rewards/judge_quality/mean": 0.3137499988079071, "rewards/judge_quality/std": 0.1665564924478531, "rewards/total_composite/mean": 0.6892167329788208, "rewards/total_composite/std": 0.2814791202545166, "reward": 0.6892167329788208, "reward_std": 0.2814791202545166, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12210405617952347, "sampling/sampling_logp_difference/max": 2.693023681640625, "sampling/importance_sampling_ratio/min": 0.06767600029706955, "sampling/importance_sampling_ratio/mean": 1.0123001337051392, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.41281329840421677, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08019299386069179, "clip_ratio/high_max": 0.08019299386069179, "clip_ratio/region_mean": 0.08019299386069179, "reward_total_mean": 0.6892167329788208, "reward_meter_mean": 0.8535219430923462, "reward_meter_std": 0.3448048233985901, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9382025003433228, "reward_repeat_soft_std": 0.06872370839118958, "reward_judge_quality_mean": 0.3137499988079071, "reward_judge_quality_std": 0.1665564924478531, "reward_total_composite_mean": 0.6892167329788208, "reward_total_composite_std": 0.2814791202545166} {"timestamp_utc": "2026-04-13T04:57:03Z", "mode": "train", "global_step": 2621, "epoch": 0.2632847815168257, "loss": 0.0618, "grad_norm": 8.774951934814453, "learning_rate": 2.0606060606060607e-06, "num_tokens": 4869214.0, "completions/mean_length": 45.875, "completions/min_length": 42.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.875, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.9475687742233276, "rewards/meter/std": 0.02803124114871025, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9202433824539185, "rewards/repeat_soft/std": 0.11091405153274536, "rewards/judge_quality/mean": 0.45250001549720764, "rewards/judge_quality/std": 0.21224987506866455, "rewards/total_composite/mean": 0.8041802644729614, "rewards/total_composite/std": 0.06122969090938568, "reward": 0.8041802644729614, "reward_std": 0.061229683458805084, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08543072640895844, "sampling/sampling_logp_difference/max": 1.2575099468231201, "sampling/importance_sampling_ratio/min": 0.28436121344566345, "sampling/importance_sampling_ratio/mean": 0.9942703247070312, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.28246392495930195, "clip_ratio/low_mean": 0.029854032676666975, "clip_ratio/low_min": 0.029854032676666975, "clip_ratio/high_mean": 0.04359802510589361, "clip_ratio/high_max": 0.04359802510589361, "clip_ratio/region_mean": 0.07345205778256059, "reward_total_mean": 0.8041802644729614, "reward_meter_mean": 0.9475687742233276, "reward_meter_std": 0.02803124114871025, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9202433824539185, "reward_repeat_soft_std": 0.11091405153274536, "reward_judge_quality_mean": 0.45250001549720764, "reward_judge_quality_std": 0.21224987506866455, "reward_total_composite_mean": 0.8041802644729614, "reward_total_composite_std": 0.06122969090938568} {"timestamp_utc": "2026-04-13T04:57:10Z", "mode": "train", "global_step": 2622, "epoch": 0.2633852335509794, "loss": 0.0248, "grad_norm": 7.7666449546813965, "learning_rate": 2.0575757575757576e-06, "num_tokens": 4871238.0, "completions/mean_length": 96.0, "completions/min_length": 87.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.0, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.7615869045257568, "rewards/meter/std": 0.3261186480522156, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9756984710693359, "rewards/repeat_soft/std": 0.023998789489269257, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7162839770317078, "rewards/total_composite/std": 0.14805929362773895, "reward": 0.7162839770317078, "reward_std": 0.14805929362773895, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11401074379682541, "sampling/sampling_logp_difference/max": 1.6168526411056519, "sampling/importance_sampling_ratio/min": 0.19852253794670105, "sampling/importance_sampling_ratio/mean": 1.0050668716430664, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5802133083343506, "clip_ratio/low_mean": 0.049719889648258686, "clip_ratio/low_min": 0.049719889648258686, "clip_ratio/high_mean": 0.06828327849507332, "clip_ratio/high_max": 0.06828327849507332, "clip_ratio/region_mean": 0.118003168143332, "reward_total_mean": 0.7162839770317078, "reward_meter_mean": 0.7615869045257568, "reward_meter_std": 0.3261186480522156, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9756984710693359, "reward_repeat_soft_std": 0.023998789489269257, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7162839770317078, "reward_total_composite_std": 0.14805929362773895} {"timestamp_utc": "2026-04-13T04:57:16Z", "mode": "train", "global_step": 2623, "epoch": 0.2634856855851331, "loss": 0.0167, "grad_norm": 11.832908630371094, "learning_rate": 2.054545454545455e-06, "num_tokens": 4872939.0, "completions/mean_length": 51.625, "completions/min_length": 45.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.625, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9809601306915283, "rewards/meter/std": 0.02066114917397499, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9826552271842957, "rewards/repeat_soft/std": 0.016987739130854607, "rewards/judge_quality/mean": 0.5012500286102295, "rewards/judge_quality/std": 0.16974246501922607, "rewards/total_composite/mean": 0.840072512626648, "rewards/total_composite/std": 0.04214463010430336, "reward": 0.840072512626648, "reward_std": 0.04214463010430336, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10829588770866394, "sampling/sampling_logp_difference/max": 2.398000717163086, "sampling/importance_sampling_ratio/min": 0.0908995047211647, "sampling/importance_sampling_ratio/mean": 0.9854894876480103, "sampling/importance_sampling_ratio/max": 1.9514236450195312, "entropy": 0.48390132561326027, "clip_ratio/low_mean": 0.08240895392373204, "clip_ratio/low_min": 0.08240895392373204, "clip_ratio/high_mean": 0.009615384973585606, "clip_ratio/high_max": 0.009615384973585606, "clip_ratio/region_mean": 0.09202433889731765, "reward_total_mean": 0.840072512626648, "reward_meter_mean": 0.9809601306915283, "reward_meter_std": 0.02066114917397499, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9826552271842957, "reward_repeat_soft_std": 0.016987739130854607, "reward_judge_quality_mean": 0.5012500286102295, "reward_judge_quality_std": 0.16974246501922607, "reward_total_composite_mean": 0.840072512626648, "reward_total_composite_std": 0.04214463010430336} {"timestamp_utc": "2026-04-13T04:57:23Z", "mode": "train", "global_step": 2624, "epoch": 0.2635861376192868, "loss": 0.0185, "grad_norm": 5.731329441070557, "learning_rate": 2.0515151515151517e-06, "num_tokens": 4875076.0, "completions/mean_length": 98.125, "completions/min_length": 87.0, "completions/max_length": 113.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.125, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.992033839225769, "rewards/meter/std": 0.0036740340292453766, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9590375423431396, "rewards/repeat_soft/std": 0.026561250910162926, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8194440007209778, "rewards/total_composite/std": 0.005009111016988754, "reward": 0.8194440007209778, "reward_std": 0.005009111016988754, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.104897141456604, "sampling/sampling_logp_difference/max": 1.60493803024292, "sampling/importance_sampling_ratio/min": 0.20090200006961823, "sampling/importance_sampling_ratio/mean": 1.0012527704238892, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5777708180248737, "clip_ratio/low_mean": 0.053904163651168346, "clip_ratio/low_min": 0.053904163651168346, "clip_ratio/high_mean": 0.05721432529389858, "clip_ratio/high_max": 0.05721432529389858, "clip_ratio/region_mean": 0.11111848894506693, "reward_total_mean": 0.8194440007209778, "reward_meter_mean": 0.992033839225769, "reward_meter_std": 0.0036740340292453766, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9590375423431396, "reward_repeat_soft_std": 0.026561250910162926, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8194440007209778, "reward_total_composite_std": 0.005009111016988754} {"timestamp_utc": "2026-04-13T04:57:30Z", "mode": "train", "global_step": 2625, "epoch": 0.2636865896534405, "loss": 0.015, "grad_norm": 15.095437049865723, "learning_rate": 2.0484848484848485e-06, "num_tokens": 4876488.0, "completions/mean_length": 40.5, "completions/min_length": 36.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.7434054017066956, "rewards/meter/std": 0.23522745072841644, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9950413703918457, "rewards/repeat_soft/std": 0.010605217888951302, "rewards/judge_quality/mean": 0.5012500286102295, "rewards/judge_quality/std": 0.16974246501922607, "rewards/total_composite/mean": 0.7344115972518921, "rewards/total_composite/std": 0.12944574654102325, "reward": 0.7344115972518921, "reward_std": 0.12944576144218445, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11375726014375687, "sampling/sampling_logp_difference/max": 1.7754542827606201, "sampling/importance_sampling_ratio/min": 0.1694064736366272, "sampling/importance_sampling_ratio/mean": 1.0022855997085571, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5087811350822449, "clip_ratio/low_mean": 0.054903456242755055, "clip_ratio/low_min": 0.054903456242755055, "clip_ratio/high_mean": 0.07167083211243153, "clip_ratio/high_max": 0.07167083211243153, "clip_ratio/region_mean": 0.12657428835518658, "reward_total_mean": 0.7344115972518921, "reward_meter_mean": 0.7434054017066956, "reward_meter_std": 0.23522745072841644, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9950413703918457, "reward_repeat_soft_std": 0.010605217888951302, "reward_judge_quality_mean": 0.5012500286102295, "reward_judge_quality_std": 0.16974246501922607, "reward_total_composite_mean": 0.7344115972518921, "reward_total_composite_std": 0.12944574654102325} {"timestamp_utc": "2026-04-13T04:57:35Z", "mode": "train", "global_step": 2626, "epoch": 0.26378704168759415, "loss": 0.03, "grad_norm": 19.863773345947266, "learning_rate": 2.0454545454545457e-06, "num_tokens": 4877928.0, "completions/mean_length": 28.0, "completions/min_length": 24.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.0, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.972572386264801, "rewards/meter/std": 0.01690923608839512, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9551723003387451, "rewards/repeat_soft/std": 0.014446047134697437, "rewards/judge_quality/mean": 0.3174999952316284, "rewards/judge_quality/std": 0.1316651552915573, "rewards/total_composite/mean": 0.7784247994422913, "rewards/total_composite/std": 0.03920810669660568, "reward": 0.7784247994422913, "reward_std": 0.03920810669660568, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12520195543766022, "sampling/sampling_logp_difference/max": 1.6617534160614014, "sampling/importance_sampling_ratio/min": 0.18980588018894196, "sampling/importance_sampling_ratio/mean": 1.0034583806991577, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6815879493951797, "clip_ratio/low_mean": 0.04523296561092138, "clip_ratio/low_min": 0.04523296561092138, "clip_ratio/high_mean": 0.05190235748887062, "clip_ratio/high_max": 0.05190235748887062, "clip_ratio/region_mean": 0.097135323099792, "reward_total_mean": 0.7784247994422913, "reward_meter_mean": 0.972572386264801, "reward_meter_std": 0.01690923608839512, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9551723003387451, "reward_repeat_soft_std": 0.014446047134697437, "reward_judge_quality_mean": 0.3174999952316284, "reward_judge_quality_std": 0.1316651552915573, "reward_total_composite_mean": 0.7784247994422913, "reward_total_composite_std": 0.03920810669660568} {"timestamp_utc": "2026-04-13T04:57:47Z", "mode": "train", "global_step": 2627, "epoch": 0.26388749372174786, "loss": -0.1235, "grad_norm": 1.6913260221481323, "learning_rate": 2.0424242424242426e-06, "num_tokens": 4879569.0, "completions/mean_length": 100.125, "completions/min_length": 33.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 41.28571701049805, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.8476735353469849, "rewards/meter/std": 0.29613614082336426, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9807596802711487, "rewards/repeat_soft/std": 0.020917825400829315, "rewards/judge_quality/mean": 0.39249998331069946, "rewards/judge_quality/std": 0.13905291259288788, "rewards/total_composite/mean": 0.7077003717422485, "rewards/total_composite/std": 0.2860512435436249, "reward": 0.7077003717422485, "reward_std": 0.2860512435436249, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08671990782022476, "sampling/sampling_logp_difference/max": 1.2807796001434326, "sampling/importance_sampling_ratio/min": 0.2778206467628479, "sampling/importance_sampling_ratio/mean": 1.005017876625061, "sampling/importance_sampling_ratio/max": 1.9486321210861206, "entropy": 0.33169933781027794, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08060354087501764, "clip_ratio/high_max": 0.08060354087501764, "clip_ratio/region_mean": 0.08060354087501764, "reward_total_mean": 0.7077003717422485, "reward_meter_mean": 0.8476735353469849, "reward_meter_std": 0.29613614082336426, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9807596802711487, "reward_repeat_soft_std": 0.020917825400829315, "reward_judge_quality_mean": 0.39249998331069946, "reward_judge_quality_std": 0.13905291259288788, "reward_total_composite_mean": 0.7077003717422485, "reward_total_composite_std": 0.2860512435436249} {"timestamp_utc": "2026-04-13T04:57:53Z", "mode": "train", "global_step": 2628, "epoch": 0.26398794575590157, "loss": 0.0511, "grad_norm": 9.704891204833984, "learning_rate": 2.03939393939394e-06, "num_tokens": 4880880.0, "completions/mean_length": 41.875, "completions/min_length": 34.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.875, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9775996208190918, "rewards/meter/std": 0.032505832612514496, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8081915378570557, "rewards/repeat_soft/std": 0.14669100940227509, "rewards/judge_quality/mean": 0.44875001907348633, "rewards/judge_quality/std": 0.21256513893604279, "rewards/total_composite/mean": 0.8053640127182007, "rewards/total_composite/std": 0.07343821972608566, "reward": 0.8053640127182007, "reward_std": 0.07343821227550507, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08673505485057831, "sampling/sampling_logp_difference/max": 2.091846466064453, "sampling/importance_sampling_ratio/min": 0.12345896661281586, "sampling/importance_sampling_ratio/mean": 0.9998816847801208, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2789176758378744, "clip_ratio/low_mean": 0.04984193784184754, "clip_ratio/low_min": 0.04984193784184754, "clip_ratio/high_mean": 0.022226866567507386, "clip_ratio/high_max": 0.022226866567507386, "clip_ratio/region_mean": 0.07206880440935493, "reward_total_mean": 0.8053640127182007, "reward_meter_mean": 0.9775996208190918, "reward_meter_std": 0.032505832612514496, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8081915378570557, "reward_repeat_soft_std": 0.14669100940227509, "reward_judge_quality_mean": 0.44875001907348633, "reward_judge_quality_std": 0.21256513893604279, "reward_total_composite_mean": 0.8053640127182007, "reward_total_composite_std": 0.07343821972608566} {"timestamp_utc": "2026-04-13T04:57:59Z", "mode": "train", "global_step": 2629, "epoch": 0.2640883977900553, "loss": -0.0023, "grad_norm": 6.711489200592041, "learning_rate": 2.0363636363636367e-06, "num_tokens": 4882748.0, "completions/mean_length": 65.5, "completions/min_length": 60.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.5, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9888550043106079, "rewards/meter/std": 0.0039055305533111095, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.959990382194519, "rewards/repeat_soft/std": 0.02432883530855179, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.8379838466644287, "rewards/total_composite/std": 0.051619332283735275, "reward": 0.8379838466644287, "reward_std": 0.05161932110786438, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0889061763882637, "sampling/sampling_logp_difference/max": 1.645876407623291, "sampling/importance_sampling_ratio/min": 0.1928434818983078, "sampling/importance_sampling_ratio/mean": 1.0073271989822388, "sampling/importance_sampling_ratio/max": 1.9328913688659668, "entropy": 0.43213922902941704, "clip_ratio/low_mean": 0.06345812161453068, "clip_ratio/low_min": 0.06345812161453068, "clip_ratio/high_mean": 0.014705882407724857, "clip_ratio/high_max": 0.014705882407724857, "clip_ratio/region_mean": 0.07816400402225554, "reward_total_mean": 0.8379838466644287, "reward_meter_mean": 0.9888550043106079, "reward_meter_std": 0.0039055305533111095, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.959990382194519, "reward_repeat_soft_std": 0.02432883530855179, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.8379838466644287, "reward_total_composite_std": 0.051619332283735275} {"timestamp_utc": "2026-04-13T04:58:06Z", "mode": "train", "global_step": 2630, "epoch": 0.2641888498242089, "loss": 0.0594, "grad_norm": 5.4918084144592285, "learning_rate": 2.0333333333333335e-06, "num_tokens": 4885361.0, "completions/mean_length": 145.625, "completions/min_length": 136.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 145.625, "completions/min_terminated_length": 136.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.9949929714202881, "rewards/meter/std": 0.0017101385165005922, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9134161472320557, "rewards/repeat_soft/std": 0.08320978283882141, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8150884509086609, "rewards/total_composite/std": 0.0076094865798950195, "reward": 0.8150884509086609, "reward_std": 0.007609505206346512, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08042175322771072, "sampling/sampling_logp_difference/max": 1.9086337089538574, "sampling/importance_sampling_ratio/min": 0.14828285574913025, "sampling/importance_sampling_ratio/mean": 1.0012969970703125, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4375581219792366, "clip_ratio/low_mean": 0.013651316054165363, "clip_ratio/low_min": 0.013651316054165363, "clip_ratio/high_mean": 0.06814868189394474, "clip_ratio/high_max": 0.06814868189394474, "clip_ratio/region_mean": 0.0817999979481101, "reward_total_mean": 0.8150884509086609, "reward_meter_mean": 0.9949929714202881, "reward_meter_std": 0.0017101385165005922, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9134161472320557, "reward_repeat_soft_std": 0.08320978283882141, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8150884509086609, "reward_total_composite_std": 0.0076094865798950195} {"timestamp_utc": "2026-04-13T04:58:13Z", "mode": "train", "global_step": 2631, "epoch": 0.26428930185836264, "loss": -0.0058, "grad_norm": 6.008114337921143, "learning_rate": 2.0303030303030303e-06, "num_tokens": 4887263.0, "completions/mean_length": 58.75, "completions/min_length": 54.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.75, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.985906720161438, "rewards/meter/std": 0.007353747263550758, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9622808694839478, "rewards/repeat_soft/std": 0.04914667084813118, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.8391361236572266, "rewards/total_composite/std": 0.04745546355843544, "reward": 0.8391361236572266, "reward_std": 0.04745547100901604, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07457123696804047, "sampling/sampling_logp_difference/max": 1.511723279953003, "sampling/importance_sampling_ratio/min": 0.22052961587905884, "sampling/importance_sampling_ratio/mean": 1.0028178691864014, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36646051332354546, "clip_ratio/low_mean": 0.059806855861097574, "clip_ratio/low_min": 0.059806855861097574, "clip_ratio/high_mean": 0.010080644860863686, "clip_ratio/high_max": 0.010080644860863686, "clip_ratio/region_mean": 0.06988750072196126, "reward_total_mean": 0.8391361236572266, "reward_meter_mean": 0.985906720161438, "reward_meter_std": 0.007353747263550758, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9622808694839478, "reward_repeat_soft_std": 0.04914667084813118, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.8391361236572266, "reward_total_composite_std": 0.04745546355843544} {"timestamp_utc": "2026-04-13T04:58:20Z", "mode": "train", "global_step": 2632, "epoch": 0.26438975389251634, "loss": -0.039, "grad_norm": 6.838648796081543, "learning_rate": 2.0272727272727276e-06, "num_tokens": 4889556.0, "completions/mean_length": 112.625, "completions/min_length": 99.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.625, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.836322009563446, "rewards/meter/std": 0.22081707417964935, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9553674459457397, "rewards/repeat_soft/std": 0.026108482852578163, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.7703816294670105, "rewards/total_composite/std": 0.12200035154819489, "reward": 0.7703816294670105, "reward_std": 0.12200034409761429, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11783672124147415, "sampling/sampling_logp_difference/max": 2.223633050918579, "sampling/importance_sampling_ratio/min": 0.10821525007486343, "sampling/importance_sampling_ratio/mean": 0.9888525009155273, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48328976333141327, "clip_ratio/low_mean": 0.026069519110023975, "clip_ratio/low_min": 0.026069519110023975, "clip_ratio/high_mean": 0.09431348647922277, "clip_ratio/high_max": 0.09431348647922277, "clip_ratio/region_mean": 0.12038300558924675, "reward_total_mean": 0.7703816294670105, "reward_meter_mean": 0.836322009563446, "reward_meter_std": 0.22081707417964935, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9553674459457397, "reward_repeat_soft_std": 0.026108482852578163, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.7703816294670105, "reward_total_composite_std": 0.12200035154819489} {"timestamp_utc": "2026-04-13T04:58:27Z", "mode": "train", "global_step": 2633, "epoch": 0.26449020592667, "loss": 0.0455, "grad_norm": 9.594964981079102, "learning_rate": 2.0242424242424244e-06, "num_tokens": 4891795.0, "completions/mean_length": 104.875, "completions/min_length": 94.0, "completions/max_length": 115.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.875, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 115.0, "rewards/meter/mean": 0.9008674025535583, "rewards/meter/std": 0.199763223528862, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.951148271560669, "rewards/repeat_soft/std": 0.022362183779478073, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.20469054579734802, "rewards/total_composite/mean": 0.8001301884651184, "rewards/total_composite/std": 0.11768932640552521, "reward": 0.8001301884651184, "reward_std": 0.11768932640552521, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11258475482463837, "sampling/sampling_logp_difference/max": 4.6792826652526855, "sampling/importance_sampling_ratio/min": 0.009285672567784786, "sampling/importance_sampling_ratio/mean": 1.0070905685424805, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48189470916986465, "clip_ratio/low_mean": 0.03134280722588301, "clip_ratio/low_min": 0.03134280722588301, "clip_ratio/high_mean": 0.0632387176156044, "clip_ratio/high_max": 0.0632387176156044, "clip_ratio/region_mean": 0.09458152484148741, "reward_total_mean": 0.8001301884651184, "reward_meter_mean": 0.9008674025535583, "reward_meter_std": 0.199763223528862, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.951148271560669, "reward_repeat_soft_std": 0.022362183779478073, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.20469054579734802, "reward_total_composite_mean": 0.8001301884651184, "reward_total_composite_std": 0.11768932640552521} {"timestamp_utc": "2026-04-13T04:58:34Z", "mode": "train", "global_step": 2634, "epoch": 0.2645906579608237, "loss": 0.0009, "grad_norm": 5.560244560241699, "learning_rate": 2.0212121212121212e-06, "num_tokens": 4894341.0, "completions/mean_length": 142.25, "completions/min_length": 126.0, "completions/max_length": 150.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.25, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 150.0, "rewards/meter/mean": 0.9938387870788574, "rewards/meter/std": 0.0021877975668758154, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9190731048583984, "rewards/repeat_soft/std": 0.03534657508134842, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.8040722608566284, "rewards/total_composite/std": 0.020474916324019432, "reward": 0.8040722608566284, "reward_std": 0.02047491818666458, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09068603068590164, "sampling/sampling_logp_difference/max": 1.8203814029693604, "sampling/importance_sampling_ratio/min": 0.16899189352989197, "sampling/importance_sampling_ratio/mean": 1.0024601221084595, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43493714556097984, "clip_ratio/low_mean": 0.01741926744580269, "clip_ratio/low_min": 0.01741926744580269, "clip_ratio/high_mean": 0.06730353040620685, "clip_ratio/high_max": 0.06730353040620685, "clip_ratio/region_mean": 0.08472279785200953, "reward_total_mean": 0.8040722608566284, "reward_meter_mean": 0.9938387870788574, "reward_meter_std": 0.0021877975668758154, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9190731048583984, "reward_repeat_soft_std": 0.03534657508134842, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.8040722608566284, "reward_total_composite_std": 0.020474916324019432} {"timestamp_utc": "2026-04-13T04:58:41Z", "mode": "train", "global_step": 2635, "epoch": 0.2646911099949774, "loss": 0.0256, "grad_norm": 6.300123691558838, "learning_rate": 2.0181818181818185e-06, "num_tokens": 4897016.0, "completions/mean_length": 141.375, "completions/min_length": 136.0, "completions/max_length": 147.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 141.375, "completions/min_terminated_length": 136.0, "completions/max_terminated_length": 147.0, "rewards/meter/mean": 0.9867814779281616, "rewards/meter/std": 0.01066283043473959, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8228534460067749, "rewards/repeat_soft/std": 0.07002797722816467, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7959619760513306, "rewards/total_composite/std": 0.020086893811821938, "reward": 0.7959619760513306, "reward_std": 0.020086882635951042, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08525662124156952, "sampling/sampling_logp_difference/max": 1.9394636154174805, "sampling/importance_sampling_ratio/min": 0.14378105103969574, "sampling/importance_sampling_ratio/mean": 1.0019174814224243, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3997477889060974, "clip_ratio/low_mean": 0.015021135564893484, "clip_ratio/low_min": 0.015021135564893484, "clip_ratio/high_mean": 0.06367240520194173, "clip_ratio/high_max": 0.06367240520194173, "clip_ratio/region_mean": 0.07869354076683521, "reward_total_mean": 0.7959619760513306, "reward_meter_mean": 0.9867814779281616, "reward_meter_std": 0.01066283043473959, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8228534460067749, "reward_repeat_soft_std": 0.07002797722816467, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7959619760513306, "reward_total_composite_std": 0.020086893811821938} {"timestamp_utc": "2026-04-13T04:58:47Z", "mode": "train", "global_step": 2636, "epoch": 0.26479156202913107, "loss": 0.0865, "grad_norm": 21.05970573425293, "learning_rate": 2.0151515151515153e-06, "num_tokens": 4898333.0, "completions/mean_length": 21.625, "completions/min_length": 18.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.625, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.9743548631668091, "rewards/meter/std": 0.020627491176128387, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9584389925003052, "rewards/repeat_soft/std": 0.011486133560538292, "rewards/judge_quality/mean": 0.44624999165534973, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8181785941123962, "rewards/total_composite/std": 0.009314045310020447, "reward": 0.8181785941123962, "reward_std": 0.009314039722084999, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12491238862276077, "sampling/sampling_logp_difference/max": 4.5922956466674805, "sampling/importance_sampling_ratio/min": 0.010129577480256557, "sampling/importance_sampling_ratio/mean": 0.9947940111160278, "sampling/importance_sampling_ratio/max": 1.6330387592315674, "entropy": 0.517784621566534, "clip_ratio/low_mean": 0.02578671369701624, "clip_ratio/low_min": 0.02578671369701624, "clip_ratio/high_mean": 0.08781518321484327, "clip_ratio/high_max": 0.08781518321484327, "clip_ratio/region_mean": 0.11360189691185951, "reward_total_mean": 0.8181785941123962, "reward_meter_mean": 0.9743548631668091, "reward_meter_std": 0.020627491176128387, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9584389925003052, "reward_repeat_soft_std": 0.011486133560538292, "reward_judge_quality_mean": 0.44624999165534973, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8181785941123962, "reward_total_composite_std": 0.009314045310020447} {"timestamp_utc": "2026-04-13T04:58:54Z", "mode": "train", "global_step": 2637, "epoch": 0.2648920140632848, "loss": -0.0118, "grad_norm": 7.0653910636901855, "learning_rate": 2.012121212121212e-06, "num_tokens": 4900159.0, "completions/mean_length": 70.25, "completions/min_length": 66.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.25, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9889947175979614, "rewards/meter/std": 0.0022520197089761496, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8756905794143677, "rewards/repeat_soft/std": 0.07606358081102371, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.809741735458374, "rewards/total_composite/std": 0.008698954246938229, "reward": 0.809741735458374, "reward_std": 0.008698969148099422, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06722798198461533, "sampling/sampling_logp_difference/max": 1.8865712881088257, "sampling/importance_sampling_ratio/min": 0.15159069001674652, "sampling/importance_sampling_ratio/mean": 1.0051085948944092, "sampling/importance_sampling_ratio/max": 1.9590650796890259, "entropy": 0.32965245097875595, "clip_ratio/low_mean": 0.026789268478751183, "clip_ratio/low_min": 0.026789268478751183, "clip_ratio/high_mean": 0.0467765397625044, "clip_ratio/high_max": 0.0467765397625044, "clip_ratio/region_mean": 0.07356580824125558, "reward_total_mean": 0.809741735458374, "reward_meter_mean": 0.9889947175979614, "reward_meter_std": 0.0022520197089761496, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8756905794143677, "reward_repeat_soft_std": 0.07606358081102371, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.809741735458374, "reward_total_composite_std": 0.008698954246938229} {"timestamp_utc": "2026-04-13T04:59:02Z", "mode": "train", "global_step": 2638, "epoch": 0.2649924660974385, "loss": 0.0992, "grad_norm": 10.084006309509277, "learning_rate": 2.009090909090909e-06, "num_tokens": 4903068.0, "completions/mean_length": 171.625, "completions/min_length": 142.0, "completions/max_length": 196.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 171.625, "completions/min_terminated_length": 142.0, "completions/max_terminated_length": 196.0, "rewards/meter/mean": 0.9852336645126343, "rewards/meter/std": 0.01456182636320591, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8420165777206421, "rewards/repeat_soft/std": 0.0649186223745346, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7821817994117737, "rewards/total_composite/std": 0.030050642788410187, "reward": 0.7821817994117737, "reward_std": 0.030050640925765038, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07771497219800949, "sampling/sampling_logp_difference/max": 2.320995807647705, "sampling/importance_sampling_ratio/min": 0.0981757715344429, "sampling/importance_sampling_ratio/mean": 1.0052087306976318, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3015572167932987, "clip_ratio/low_mean": 0.02425468945875764, "clip_ratio/low_min": 0.02425468945875764, "clip_ratio/high_mean": 0.040393671952188015, "clip_ratio/high_max": 0.040393671952188015, "clip_ratio/region_mean": 0.06464836141094565, "reward_total_mean": 0.7821817994117737, "reward_meter_mean": 0.9852336645126343, "reward_meter_std": 0.01456182636320591, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8420165777206421, "reward_repeat_soft_std": 0.0649186223745346, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7821817994117737, "reward_total_composite_std": 0.030050642788410187} {"timestamp_utc": "2026-04-13T04:59:08Z", "mode": "train", "global_step": 2639, "epoch": 0.2650929181315922, "loss": -0.0038, "grad_norm": 11.236026763916016, "learning_rate": 2.0060606060606062e-06, "num_tokens": 4904529.0, "completions/mean_length": 31.625, "completions/min_length": 30.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.625, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.6897780895233154, "rewards/meter/std": 0.4232070744037628, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.9312499761581421, "rewards/judge_quality/std": 0.015526460483670235, "rewards/total_composite/mean": 0.8360251188278198, "rewards/total_composite/std": 0.19299280643463135, "reward": 0.8360251188278198, "reward_std": 0.19299279153347015, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09830275177955627, "sampling/sampling_logp_difference/max": 1.7749881744384766, "sampling/importance_sampling_ratio/min": 0.1694854497909546, "sampling/importance_sampling_ratio/mean": 1.0101711750030518, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47293607890605927, "clip_ratio/low_mean": 0.044657258316874504, "clip_ratio/low_min": 0.044657258316874504, "clip_ratio/high_mean": 0.05919180647470057, "clip_ratio/high_max": 0.05919180647470057, "clip_ratio/region_mean": 0.10384906479157507, "reward_total_mean": 0.8360251188278198, "reward_meter_mean": 0.6897780895233154, "reward_meter_std": 0.4232070744037628, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.9312499761581421, "reward_judge_quality_std": 0.015526460483670235, "reward_total_composite_mean": 0.8360251188278198, "reward_total_composite_std": 0.19299280643463135} {"timestamp_utc": "2026-04-13T04:59:14Z", "mode": "train", "global_step": 2640, "epoch": 0.26519337016574585, "loss": 0.005, "grad_norm": 11.370562553405762, "learning_rate": 2.0030303030303035e-06, "num_tokens": 4905957.0, "completions/mean_length": 25.5, "completions/min_length": 23.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.5, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9907851219177246, "rewards/meter/std": 0.0051386067643761635, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9500192403793335, "rewards/repeat_soft/std": 0.022354967892169952, "rewards/judge_quality/mean": 0.5049999952316284, "rewards/judge_quality/std": 0.16801361739635468, "rewards/total_composite/mean": 0.8423552513122559, "rewards/total_composite/std": 0.04827903211116791, "reward": 0.8423552513122559, "reward_std": 0.048279035836458206, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08186929672956467, "sampling/sampling_logp_difference/max": 1.5496716499328613, "sampling/importance_sampling_ratio/min": 0.21231767535209656, "sampling/importance_sampling_ratio/mean": 0.9912213087081909, "sampling/importance_sampling_ratio/max": 1.8431888818740845, "entropy": 0.407336987555027, "clip_ratio/low_mean": 0.05401409789919853, "clip_ratio/low_min": 0.05401409789919853, "clip_ratio/high_mean": 0.009615384973585606, "clip_ratio/high_max": 0.009615384973585606, "clip_ratio/region_mean": 0.06362948287278414, "reward_total_mean": 0.8423552513122559, "reward_meter_mean": 0.9907851219177246, "reward_meter_std": 0.0051386067643761635, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9500192403793335, "reward_repeat_soft_std": 0.022354967892169952, "reward_judge_quality_mean": 0.5049999952316284, "reward_judge_quality_std": 0.16801361739635468, "reward_total_composite_mean": 0.8423552513122559, "reward_total_composite_std": 0.04827903211116791} {"timestamp_utc": "2026-04-13T04:59:25Z", "mode": "train", "global_step": 2641, "epoch": 0.26529382219989955, "loss": -0.1464, "grad_norm": 1.613852858543396, "learning_rate": 2.0000000000000003e-06, "num_tokens": 4907960.0, "completions/mean_length": 180.375, "completions/min_length": 67.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 69.83333587646484, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.6937543153762817, "rewards/meter/std": 0.34672313928604126, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9921072721481323, "rewards/repeat_soft/std": 0.013384324498474598, "rewards/judge_quality/mean": 0.45250001549720764, "rewards/judge_quality/std": 0.33065950870513916, "rewards/total_composite/mean": 0.5938628315925598, "rewards/total_composite/std": 0.3811085522174835, "reward": 0.5938628315925598, "reward_std": 0.3811085522174835, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09827672690153122, "sampling/sampling_logp_difference/max": 1.4094386100769043, "sampling/importance_sampling_ratio/min": 0.2442803978919983, "sampling/importance_sampling_ratio/mean": 1.0080251693725586, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4520026259124279, "clip_ratio/low_mean": 0.0086805559694767, "clip_ratio/low_min": 0.0086805559694767, "clip_ratio/high_mean": 0.06797636393457651, "clip_ratio/high_max": 0.06797636393457651, "clip_ratio/region_mean": 0.07665691990405321, "reward_total_mean": 0.5938628315925598, "reward_meter_mean": 0.6937543153762817, "reward_meter_std": 0.34672313928604126, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9921072721481323, "reward_repeat_soft_std": 0.013384324498474598, "reward_judge_quality_mean": 0.45250001549720764, "reward_judge_quality_std": 0.33065950870513916, "reward_total_composite_mean": 0.5938628315925598, "reward_total_composite_std": 0.3811085522174835} {"timestamp_utc": "2026-04-13T04:59:33Z", "mode": "train", "global_step": 2642, "epoch": 0.26539427423405326, "loss": 0.0155, "grad_norm": 4.214881896972656, "learning_rate": 1.996969696969697e-06, "num_tokens": 4910607.0, "completions/mean_length": 146.875, "completions/min_length": 138.0, "completions/max_length": 159.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 146.875, "completions/min_terminated_length": 138.0, "completions/max_terminated_length": 159.0, "rewards/meter/mean": 0.9243253469467163, "rewards/meter/std": 0.18180710077285767, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7055773735046387, "rewards/repeat_soft/std": 0.10821584612131119, "rewards/judge_quality/mean": 0.565000057220459, "rewards/judge_quality/std": 0.3022770881652832, "rewards/total_composite/mean": 0.8060041666030884, "rewards/total_composite/std": 0.08069545775651932, "reward": 0.8060041666030884, "reward_std": 0.08069546520709991, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07690355181694031, "sampling/sampling_logp_difference/max": 2.969057559967041, "sampling/importance_sampling_ratio/min": 0.05135168507695198, "sampling/importance_sampling_ratio/mean": 1.001613736152649, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3154064267873764, "clip_ratio/low_mean": 0.05115943029522896, "clip_ratio/low_min": 0.05115943029522896, "clip_ratio/high_mean": 0.013136007823050022, "clip_ratio/high_max": 0.013136007823050022, "clip_ratio/region_mean": 0.06429543811827898, "reward_total_mean": 0.8060041666030884, "reward_meter_mean": 0.9243253469467163, "reward_meter_std": 0.18180710077285767, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7055773735046387, "reward_repeat_soft_std": 0.10821584612131119, "reward_judge_quality_mean": 0.565000057220459, "reward_judge_quality_std": 0.3022770881652832, "reward_total_composite_mean": 0.8060041666030884, "reward_total_composite_std": 0.08069545775651932} {"timestamp_utc": "2026-04-13T04:59:40Z", "mode": "train", "global_step": 2643, "epoch": 0.2654947262682069, "loss": 0.0695, "grad_norm": 8.113642692565918, "learning_rate": 1.993939393939394e-06, "num_tokens": 4912488.0, "completions/mean_length": 63.125, "completions/min_length": 56.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.125, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9918325543403625, "rewards/meter/std": 0.001990068703889847, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.971135139465332, "rewards/repeat_soft/std": 0.014020724222064018, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.821688175201416, "rewards/total_composite/std": 0.00504273921251297, "reward": 0.821688175201416, "reward_std": 0.005042738281190395, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10435120761394501, "sampling/sampling_logp_difference/max": 1.1688480377197266, "sampling/importance_sampling_ratio/min": 0.310724675655365, "sampling/importance_sampling_ratio/mean": 1.0084868669509888, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6091186217963696, "clip_ratio/low_mean": 0.08088625222444534, "clip_ratio/low_min": 0.08088625222444534, "clip_ratio/high_mean": 0.030152225866913795, "clip_ratio/high_max": 0.030152225866913795, "clip_ratio/region_mean": 0.11103847809135914, "reward_total_mean": 0.821688175201416, "reward_meter_mean": 0.9918325543403625, "reward_meter_std": 0.001990068703889847, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.971135139465332, "reward_repeat_soft_std": 0.014020724222064018, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.821688175201416, "reward_total_composite_std": 0.00504273921251297} {"timestamp_utc": "2026-04-13T04:59:51Z", "mode": "train", "global_step": 2644, "epoch": 0.2655951783023606, "loss": -0.0982, "grad_norm": 1.632641315460205, "learning_rate": 1.9909090909090913e-06, "num_tokens": 4913924.0, "completions/mean_length": 89.5, "completions/min_length": 27.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 29.142858505249023, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.985951840877533, "rewards/meter/std": 0.01737314835190773, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9496357440948486, "rewards/repeat_soft/std": 0.03638550266623497, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.13617216050624847, "rewards/total_composite/mean": 0.7183141708374023, "rewards/total_composite/std": 0.2903323471546173, "reward": 0.7183141708374023, "reward_std": 0.2903323471546173, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10215900093317032, "sampling/sampling_logp_difference/max": 2.880985736846924, "sampling/importance_sampling_ratio/min": 0.056079454720020294, "sampling/importance_sampling_ratio/mean": 1.012040376663208, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4879857823252678, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08584302989766002, "clip_ratio/high_max": 0.08584302989766002, "clip_ratio/region_mean": 0.08584302989766002, "reward_total_mean": 0.7183141708374023, "reward_meter_mean": 0.985951840877533, "reward_meter_std": 0.01737314835190773, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9496357440948486, "reward_repeat_soft_std": 0.03638550266623497, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.13617216050624847, "reward_total_composite_mean": 0.7183141708374023, "reward_total_composite_std": 0.2903323471546173} {"timestamp_utc": "2026-04-13T04:59:58Z", "mode": "train", "global_step": 2645, "epoch": 0.26569563033651433, "loss": 0.0315, "grad_norm": 9.930087089538574, "learning_rate": 1.987878787878788e-06, "num_tokens": 4915736.0, "completions/mean_length": 52.5, "completions/min_length": 49.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.5, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9345169067382812, "rewards/meter/std": 0.11698256433010101, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.970736563205719, "rewards/repeat_soft/std": 0.020550735294818878, "rewards/judge_quality/mean": 0.3687499761581421, "rewards/judge_quality/std": 0.10802611708641052, "rewards/total_composite/mean": 0.7782312035560608, "rewards/total_composite/std": 0.05392498895525932, "reward": 0.7782312035560608, "reward_std": 0.053924985229969025, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10667314380407333, "sampling/sampling_logp_difference/max": 1.396815299987793, "sampling/importance_sampling_ratio/min": 0.24738354980945587, "sampling/importance_sampling_ratio/mean": 1.0115571022033691, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5085925497114658, "clip_ratio/low_mean": 0.016392805613577366, "clip_ratio/low_min": 0.016392805613577366, "clip_ratio/high_mean": 0.07248423947021365, "clip_ratio/high_max": 0.07248423947021365, "clip_ratio/region_mean": 0.08887704508379102, "reward_total_mean": 0.7782312035560608, "reward_meter_mean": 0.9345169067382812, "reward_meter_std": 0.11698256433010101, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.970736563205719, "reward_repeat_soft_std": 0.020550735294818878, "reward_judge_quality_mean": 0.3687499761581421, "reward_judge_quality_std": 0.10802611708641052, "reward_total_composite_mean": 0.7782312035560608, "reward_total_composite_std": 0.05392498895525932} {"timestamp_utc": "2026-04-13T05:00:06Z", "mode": "train", "global_step": 2646, "epoch": 0.265796082370668, "loss": 0.0332, "grad_norm": 6.800406455993652, "learning_rate": 1.984848484848485e-06, "num_tokens": 4918175.0, "completions/mean_length": 114.875, "completions/min_length": 109.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.875, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.9662849307060242, "rewards/meter/std": 0.024538768455386162, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7756881713867188, "rewards/repeat_soft/std": 0.05197707563638687, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7820219993591309, "rewards/total_composite/std": 0.018233338370919228, "reward": 0.7820219993591309, "reward_std": 0.018233338370919228, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06681504845619202, "sampling/sampling_logp_difference/max": 1.6335296630859375, "sampling/importance_sampling_ratio/min": 0.19523923099040985, "sampling/importance_sampling_ratio/mean": 1.0045338869094849, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.26984189450740814, "clip_ratio/low_mean": 0.023692018818110228, "clip_ratio/low_min": 0.023692018818110228, "clip_ratio/high_mean": 0.03406417788937688, "clip_ratio/high_max": 0.03406417788937688, "clip_ratio/region_mean": 0.057756196707487106, "reward_total_mean": 0.7820219993591309, "reward_meter_mean": 0.9662849307060242, "reward_meter_std": 0.024538768455386162, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7756881713867188, "reward_repeat_soft_std": 0.05197707563638687, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7820219993591309, "reward_total_composite_std": 0.018233338370919228} {"timestamp_utc": "2026-04-13T05:00:13Z", "mode": "train", "global_step": 2647, "epoch": 0.2658965344048217, "loss": 0.002, "grad_norm": 9.173641204833984, "learning_rate": 1.981818181818182e-06, "num_tokens": 4920128.0, "completions/mean_length": 72.125, "completions/min_length": 61.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.125, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.9621886610984802, "rewards/meter/std": 0.06331313401460648, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8579273819923401, "rewards/repeat_soft/std": 0.024629296734929085, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7947776317596436, "rewards/total_composite/std": 0.02938523143529892, "reward": 0.7947776317596436, "reward_std": 0.029385237023234367, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07970039546489716, "sampling/sampling_logp_difference/max": 1.9714994430541992, "sampling/importance_sampling_ratio/min": 0.13924790918827057, "sampling/importance_sampling_ratio/mean": 1.0036509037017822, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.29596488550305367, "clip_ratio/low_mean": 0.007692307699471712, "clip_ratio/low_min": 0.007692307699471712, "clip_ratio/high_mean": 0.07555073406547308, "clip_ratio/high_max": 0.07555073406547308, "clip_ratio/region_mean": 0.08324304176494479, "reward_total_mean": 0.7947776317596436, "reward_meter_mean": 0.9621886610984802, "reward_meter_std": 0.06331313401460648, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8579273819923401, "reward_repeat_soft_std": 0.024629296734929085, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7947776317596436, "reward_total_composite_std": 0.02938523143529892} {"timestamp_utc": "2026-04-13T05:00:20Z", "mode": "train", "global_step": 2648, "epoch": 0.2659969864389754, "loss": 0.0041, "grad_norm": 14.678553581237793, "learning_rate": 1.978787878787879e-06, "num_tokens": 4921402.0, "completions/mean_length": 25.25, "completions/min_length": 22.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.25, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9670838713645935, "rewards/meter/std": 0.06029244884848595, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9560075998306274, "rewards/repeat_soft/std": 0.012021517381072044, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404788017273, "rewards/total_composite/mean": 0.8476635217666626, "rewards/total_composite/std": 0.07772732526063919, "reward": 0.8476635217666626, "reward_std": 0.077727310359478, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12589187920093536, "sampling/sampling_logp_difference/max": 1.4561638832092285, "sampling/importance_sampling_ratio/min": 0.2331288754940033, "sampling/importance_sampling_ratio/mean": 0.9945253729820251, "sampling/importance_sampling_ratio/max": 1.774803876876831, "entropy": 0.5894707068800926, "clip_ratio/low_mean": 0.0836351178586483, "clip_ratio/low_min": 0.0836351178586483, "clip_ratio/high_mean": 0.03365384694188833, "clip_ratio/high_max": 0.03365384694188833, "clip_ratio/region_mean": 0.11728896480053663, "reward_total_mean": 0.8476635217666626, "reward_meter_mean": 0.9670838713645935, "reward_meter_std": 0.06029244884848595, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9560075998306274, "reward_repeat_soft_std": 0.012021517381072044, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404788017273, "reward_total_composite_mean": 0.8476635217666626, "reward_total_composite_std": 0.07772732526063919} {"timestamp_utc": "2026-04-13T05:00:31Z", "mode": "train", "global_step": 2649, "epoch": 0.26609743847312906, "loss": -0.2305, "grad_norm": 1.3511441946029663, "learning_rate": 1.975757575757576e-06, "num_tokens": 4923768.0, "completions/mean_length": 195.75, "completions/min_length": 124.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 150.57144165039062, "completions/min_terminated_length": 124.0, "completions/max_terminated_length": 171.0, "rewards/meter/mean": 0.9155026078224182, "rewards/meter/std": 0.20014554262161255, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.6937511563301086, "rewards/repeat_soft/std": 0.08207808434963226, "rewards/judge_quality/mean": 0.33124998211860657, "rewards/judge_quality/std": 0.13715866208076477, "rewards/total_composite/mean": 0.6614018082618713, "rewards/total_composite/std": 0.2685467600822449, "reward": 0.6614018082618713, "reward_std": 0.26854678988456726, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.054622549563646317, "sampling/sampling_logp_difference/max": 1.5973519086837769, "sampling/importance_sampling_ratio/min": 0.20243185758590698, "sampling/importance_sampling_ratio/mean": 1.0045727491378784, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.21285836026072502, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.051013028249144554, "clip_ratio/high_max": 0.051013028249144554, "clip_ratio/region_mean": 0.051013028249144554, "reward_total_mean": 0.6614018082618713, "reward_meter_mean": 0.9155026078224182, "reward_meter_std": 0.20014554262161255, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.6937511563301086, "reward_repeat_soft_std": 0.08207808434963226, "reward_judge_quality_mean": 0.33124998211860657, "reward_judge_quality_std": 0.13715866208076477, "reward_total_composite_mean": 0.6614018082618713, "reward_total_composite_std": 0.2685467600822449} {"timestamp_utc": "2026-04-13T05:00:39Z", "mode": "train", "global_step": 2650, "epoch": 0.26619789050728276, "loss": 0.0188, "grad_norm": 8.595894813537598, "learning_rate": 1.9727272727272727e-06, "num_tokens": 4925901.0, "completions/mean_length": 103.625, "completions/min_length": 99.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.625, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.9935132265090942, "rewards/meter/std": 0.002907405374571681, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.776716947555542, "rewards/repeat_soft/std": 0.08065928518772125, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7943776845932007, "rewards/total_composite/std": 0.017417754977941513, "reward": 0.7943776845932007, "reward_std": 0.017417747527360916, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.070249043405056, "sampling/sampling_logp_difference/max": 1.3592286109924316, "sampling/importance_sampling_ratio/min": 0.25685885548591614, "sampling/importance_sampling_ratio/mean": 1.000645399093628, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3230650797486305, "clip_ratio/low_mean": 0.020701338071376085, "clip_ratio/low_min": 0.020701338071376085, "clip_ratio/high_mean": 0.03993973881006241, "clip_ratio/high_max": 0.03993973881006241, "clip_ratio/region_mean": 0.060641076881438494, "reward_total_mean": 0.7943776845932007, "reward_meter_mean": 0.9935132265090942, "reward_meter_std": 0.002907405374571681, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.776716947555542, "reward_repeat_soft_std": 0.08065928518772125, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7943776845932007, "reward_total_composite_std": 0.017417754977941513} {"timestamp_utc": "2026-04-13T05:01:27Z", "mode": "eval", "global_step": 2650, "epoch": 0.26619789050728276, "eval_loss": NaN, "eval_runtime": 48.3517, "eval_samples_per_second": 1.655, "eval_steps_per_second": 0.207, "eval_num_tokens": 4925901.0, "eval_completions/mean_length": 105.675, "eval_completions/min_length": 44.9, "eval_completions/max_length": 201.4, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 96.39583435058594, "eval_completions/min_terminated_length": 44.9, "eval_completions/max_terminated_length": 167.5, "eval_rewards/meter/mean": 0.9272557973861695, "eval_rewards/meter/std": 0.13491792539134623, "eval_rewards/count_adherence/mean": 0.9770833313465118, "eval_rewards/count_adherence/std": 0.05291357152163982, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.046291005611419675, "eval_rewards/repeat_soft/mean": 0.9016985237598419, "eval_rewards/repeat_soft/std": 0.09095696061849594, "eval_rewards/judge_quality/mean": 0.4545000046491623, "eval_rewards/judge_quality/std": 0.1446153403259814, "eval_rewards/total_composite/mean": 0.7791326224803925, "eval_rewards/total_composite/std": 0.10303807687014341, "eval_reward": 0.7791326224803925, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.03904734812676906, "eval_sampling/sampling_logp_difference/max": 0.8979687452316284, "eval_sampling/importance_sampling_ratio/min": 0.4326219066977501, "eval_sampling/importance_sampling_ratio/mean": 1.0066031575202943, "eval_sampling/importance_sampling_ratio/max": 1.3205069422721862, "eval_entropy": 0.3915812075138092, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7791326224803925, "eval_reward_meter_mean": 0.9272557973861695, "eval_reward_meter_std": 0.13491792539134623, "eval_reward_count_adherence_mean": 0.9770833313465118, "eval_reward_count_adherence_std": 0.05291357152163982, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.046291005611419675, "eval_reward_repeat_soft_mean": 0.9016985237598419, "eval_reward_repeat_soft_std": 0.09095696061849594, "eval_reward_judge_quality_mean": 0.4545000046491623, "eval_reward_judge_quality_std": 0.1446153403259814, "eval_reward_total_composite_mean": 0.7791326224803925, "eval_reward_total_composite_std": 0.10303807687014341} {"timestamp_utc": "2026-04-13T05:01:37Z", "mode": "train", "global_step": 2651, "epoch": 0.2662983425414365, "loss": -0.0121, "grad_norm": 9.759740829467773, "learning_rate": 1.96969696969697e-06, "num_tokens": 4927587.0, "completions/mean_length": 60.75, "completions/min_length": 58.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.75, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.7714345455169678, "rewards/meter/std": 0.3394203782081604, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9730754494667053, "rewards/repeat_soft/std": 0.028154416009783745, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404788017273, "rewards/total_composite/mean": 0.7613280415534973, "rewards/total_composite/std": 0.13607926666736603, "reward": 0.7613280415534973, "reward_std": 0.13607926666736603, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09103674441576004, "sampling/sampling_logp_difference/max": 1.8271520137786865, "sampling/importance_sampling_ratio/min": 0.16087107360363007, "sampling/importance_sampling_ratio/mean": 0.9936968088150024, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3775857165455818, "clip_ratio/low_mean": 0.02133255358785391, "clip_ratio/low_min": 0.02133255358785391, "clip_ratio/high_mean": 0.06105630844831467, "clip_ratio/high_max": 0.06105630844831467, "clip_ratio/region_mean": 0.08238886203616858, "reward_total_mean": 0.7613280415534973, "reward_meter_mean": 0.7714345455169678, "reward_meter_std": 0.3394203782081604, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9730754494667053, "reward_repeat_soft_std": 0.028154416009783745, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404788017273, "reward_total_composite_mean": 0.7613280415534973, "reward_total_composite_std": 0.13607926666736603} {"timestamp_utc": "2026-04-13T05:01:43Z", "mode": "train", "global_step": 2652, "epoch": 0.2663987945755902, "loss": 0.0387, "grad_norm": 10.08763313293457, "learning_rate": 1.9666666666666668e-06, "num_tokens": 4929533.0, "completions/mean_length": 58.25, "completions/min_length": 54.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.25, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.6775562763214111, "rewards/meter/std": 0.37627285718917847, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9346755743026733, "rewards/repeat_soft/std": 0.04258989542722702, "rewards/judge_quality/mean": 0.7362500429153442, "rewards/judge_quality/std": 0.25376805663108826, "rewards/total_composite/mean": 0.7692428827285767, "rewards/total_composite/std": 0.1803382784128189, "reward": 0.7692428827285767, "reward_std": 0.18033826351165771, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09821439534425735, "sampling/sampling_logp_difference/max": 1.8005690574645996, "sampling/importance_sampling_ratio/min": 0.16520485281944275, "sampling/importance_sampling_ratio/mean": 1.025367021560669, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5114798098802567, "clip_ratio/low_mean": 0.03503465000540018, "clip_ratio/low_min": 0.03503465000540018, "clip_ratio/high_mean": 0.05057596578262746, "clip_ratio/high_max": 0.05057596578262746, "clip_ratio/region_mean": 0.08561061578802764, "reward_total_mean": 0.7692428827285767, "reward_meter_mean": 0.6775562763214111, "reward_meter_std": 0.37627285718917847, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9346755743026733, "reward_repeat_soft_std": 0.04258989542722702, "reward_judge_quality_mean": 0.7362500429153442, "reward_judge_quality_std": 0.25376805663108826, "reward_total_composite_mean": 0.7692428827285767, "reward_total_composite_std": 0.1803382784128189} {"timestamp_utc": "2026-04-13T05:01:50Z", "mode": "train", "global_step": 2653, "epoch": 0.26649924660974383, "loss": 0.0229, "grad_norm": 9.052011489868164, "learning_rate": 1.9636363636363636e-06, "num_tokens": 4931262.0, "completions/mean_length": 65.125, "completions/min_length": 59.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.125, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9937130212783813, "rewards/meter/std": 0.00381490052677691, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9441758394241333, "rewards/repeat_soft/std": 0.05449923500418663, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.8010884523391724, "rewards/total_composite/std": 0.036816492676734924, "reward": 0.8010884523391724, "reward_std": 0.03681648522615433, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09954728931188583, "sampling/sampling_logp_difference/max": 1.4992964267730713, "sampling/importance_sampling_ratio/min": 0.2232872098684311, "sampling/importance_sampling_ratio/mean": 0.9988120794296265, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5109491050243378, "clip_ratio/low_mean": 0.015355417970567942, "clip_ratio/low_min": 0.015355417970567942, "clip_ratio/high_mean": 0.08023014292120934, "clip_ratio/high_max": 0.08023014292120934, "clip_ratio/region_mean": 0.09558556089177728, "reward_total_mean": 0.8010884523391724, "reward_meter_mean": 0.9937130212783813, "reward_meter_std": 0.00381490052677691, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9441758394241333, "reward_repeat_soft_std": 0.05449923500418663, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.8010884523391724, "reward_total_composite_std": 0.036816492676734924} {"timestamp_utc": "2026-04-13T05:01:57Z", "mode": "train", "global_step": 2654, "epoch": 0.26659969864389754, "loss": 0.0347, "grad_norm": 7.055185317993164, "learning_rate": 1.960606060606061e-06, "num_tokens": 4933418.0, "completions/mean_length": 95.5, "completions/min_length": 85.0, "completions/max_length": 110.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.5, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 110.0, "rewards/meter/mean": 0.9241846203804016, "rewards/meter/std": 0.02887849323451519, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7837920188903809, "rewards/repeat_soft/std": 0.21042795479297638, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7591997385025024, "rewards/total_composite/std": 0.024637693539261818, "reward": 0.7591997385025024, "reward_std": 0.024637699127197266, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07450598478317261, "sampling/sampling_logp_difference/max": 1.4864845275878906, "sampling/importance_sampling_ratio/min": 0.22616633772850037, "sampling/importance_sampling_ratio/mean": 1.0075445175170898, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31721447966992855, "clip_ratio/low_mean": 0.022631515748798847, "clip_ratio/low_min": 0.022631515748798847, "clip_ratio/high_mean": 0.053721125703305006, "clip_ratio/high_max": 0.053721125703305006, "clip_ratio/region_mean": 0.07635264145210385, "reward_total_mean": 0.7591997385025024, "reward_meter_mean": 0.9241846203804016, "reward_meter_std": 0.02887849323451519, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7837920188903809, "reward_repeat_soft_std": 0.21042795479297638, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7591997385025024, "reward_total_composite_std": 0.024637693539261818} {"timestamp_utc": "2026-04-13T05:02:04Z", "mode": "train", "global_step": 2655, "epoch": 0.26670015067805125, "loss": 0.0372, "grad_norm": 11.47807502746582, "learning_rate": 1.9575757575757577e-06, "num_tokens": 4935047.0, "completions/mean_length": 43.625, "completions/min_length": 38.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.625, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.8490097522735596, "rewards/meter/std": 0.25331369042396545, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9416637420654297, "rewards/repeat_soft/std": 0.08908527344465256, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.11444743722677231, "rewards/total_composite/mean": 0.766845703125, "rewards/total_composite/std": 0.1250019669532776, "reward": 0.766845703125, "reward_std": 0.1250019520521164, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10724533349275589, "sampling/sampling_logp_difference/max": 1.2409639358520508, "sampling/importance_sampling_ratio/min": 0.2891054153442383, "sampling/importance_sampling_ratio/mean": 0.9915753602981567, "sampling/importance_sampling_ratio/max": 1.944963812828064, "entropy": 0.4531001150608063, "clip_ratio/low_mean": 0.005681818351149559, "clip_ratio/low_min": 0.005681818351149559, "clip_ratio/high_mean": 0.1034809947013855, "clip_ratio/high_max": 0.1034809947013855, "clip_ratio/region_mean": 0.10916281305253506, "reward_total_mean": 0.766845703125, "reward_meter_mean": 0.8490097522735596, "reward_meter_std": 0.25331369042396545, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9416637420654297, "reward_repeat_soft_std": 0.08908527344465256, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.11444743722677231, "reward_total_composite_mean": 0.766845703125, "reward_total_composite_std": 0.1250019669532776} {"timestamp_utc": "2026-04-13T05:02:11Z", "mode": "train", "global_step": 2656, "epoch": 0.2668006027122049, "loss": 0.0108, "grad_norm": 11.300090789794922, "learning_rate": 1.954545454545455e-06, "num_tokens": 4937000.0, "completions/mean_length": 68.125, "completions/min_length": 64.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.125, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.8634822368621826, "rewards/meter/std": 0.11938398331403732, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9608625173568726, "rewards/repeat_soft/std": 0.022132597863674164, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.21084441244602203, "rewards/total_composite/mean": 0.7865282297134399, "rewards/total_composite/std": 0.07919927686452866, "reward": 0.7865282297134399, "reward_std": 0.07919929176568985, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11126866191625595, "sampling/sampling_logp_difference/max": 2.8834333419799805, "sampling/importance_sampling_ratio/min": 0.055942367762327194, "sampling/importance_sampling_ratio/mean": 0.9946016073226929, "sampling/importance_sampling_ratio/max": 1.9965564012527466, "entropy": 0.37536685168743134, "clip_ratio/low_mean": 0.0390625, "clip_ratio/low_min": 0.0390625, "clip_ratio/high_mean": 0.06353825610131025, "clip_ratio/high_max": 0.06353825610131025, "clip_ratio/region_mean": 0.10260075610131025, "reward_total_mean": 0.7865282297134399, "reward_meter_mean": 0.8634822368621826, "reward_meter_std": 0.11938398331403732, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9608625173568726, "reward_repeat_soft_std": 0.022132597863674164, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.21084441244602203, "reward_total_composite_mean": 0.7865282297134399, "reward_total_composite_std": 0.07919927686452866} {"timestamp_utc": "2026-04-13T05:02:17Z", "mode": "train", "global_step": 2657, "epoch": 0.2669010547463586, "loss": 0.0183, "grad_norm": 7.625934600830078, "learning_rate": 1.9515151515151518e-06, "num_tokens": 4939169.0, "completions/mean_length": 92.125, "completions/min_length": 86.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.125, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.976838231086731, "rewards/meter/std": 0.021861907094717026, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9208076000213623, "rewards/repeat_soft/std": 0.05067681893706322, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.8826580047607422, "rewards/total_composite/std": 0.08562929183244705, "reward": 0.8826580047607422, "reward_std": 0.08562927693128586, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07678907364606857, "sampling/sampling_logp_difference/max": 2.2374401092529297, "sampling/importance_sampling_ratio/min": 0.10673137754201889, "sampling/importance_sampling_ratio/mean": 0.9982582330703735, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.38951433449983597, "clip_ratio/low_mean": 0.02853392530232668, "clip_ratio/low_min": 0.02853392530232668, "clip_ratio/high_mean": 0.037766237277537584, "clip_ratio/high_max": 0.037766237277537584, "clip_ratio/region_mean": 0.06630016257986426, "reward_total_mean": 0.8826580047607422, "reward_meter_mean": 0.976838231086731, "reward_meter_std": 0.021861907094717026, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9208076000213623, "reward_repeat_soft_std": 0.05067681893706322, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.8826580047607422, "reward_total_composite_std": 0.08562929183244705} {"timestamp_utc": "2026-04-13T05:02:26Z", "mode": "train", "global_step": 2658, "epoch": 0.2670015067805123, "loss": 0.0119, "grad_norm": 5.021048545837402, "learning_rate": 1.9484848484848486e-06, "num_tokens": 4942639.0, "completions/mean_length": 204.75, "completions/min_length": 167.0, "completions/max_length": 220.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 204.75, "completions/min_terminated_length": 167.0, "completions/max_terminated_length": 220.0, "rewards/meter/mean": 0.9922095537185669, "rewards/meter/std": 0.00709010474383831, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8763144016265869, "rewards/repeat_soft/std": 0.06480276584625244, "rewards/judge_quality/mean": 0.29249998927116394, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7656257152557373, "rewards/total_composite/std": 0.025580277666449547, "reward": 0.7656257152557373, "reward_std": 0.025580277666449547, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08779054135084152, "sampling/sampling_logp_difference/max": 4.284206390380859, "sampling/importance_sampling_ratio/min": 0.013784556649625301, "sampling/importance_sampling_ratio/mean": 0.9994310736656189, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.35063719376921654, "clip_ratio/low_mean": 0.03424276039004326, "clip_ratio/low_min": 0.03424276039004326, "clip_ratio/high_mean": 0.0439658397808671, "clip_ratio/high_max": 0.0439658397808671, "clip_ratio/region_mean": 0.07820860017091036, "reward_total_mean": 0.7656257152557373, "reward_meter_mean": 0.9922095537185669, "reward_meter_std": 0.00709010474383831, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8763144016265869, "reward_repeat_soft_std": 0.06480276584625244, "reward_judge_quality_mean": 0.29249998927116394, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7656257152557373, "reward_total_composite_std": 0.025580277666449547} {"timestamp_utc": "2026-04-13T05:02:34Z", "mode": "train", "global_step": 2659, "epoch": 0.267101958814666, "loss": -0.0016, "grad_norm": 4.9795823097229, "learning_rate": 1.945454545454546e-06, "num_tokens": 4945123.0, "completions/mean_length": 122.5, "completions/min_length": 118.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.5, "completions/min_terminated_length": 118.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9477174878120422, "rewards/meter/std": 0.030998004600405693, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6746211051940918, "rewards/repeat_soft/std": 0.07521852850914001, "rewards/judge_quality/mean": 0.2212499976158142, "rewards/judge_quality/std": 0.09433034062385559, "rewards/total_composite/mean": 0.6803100109100342, "rewards/total_composite/std": 0.018086643889546394, "reward": 0.6803100109100342, "reward_std": 0.0180866327136755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0513748973608017, "sampling/sampling_logp_difference/max": 2.104152202606201, "sampling/importance_sampling_ratio/min": 0.12194901704788208, "sampling/importance_sampling_ratio/mean": 0.9966181516647339, "sampling/importance_sampling_ratio/max": 1.9115839004516602, "entropy": 0.19727863185107708, "clip_ratio/low_mean": 0.031204916071146727, "clip_ratio/low_min": 0.031204916071146727, "clip_ratio/high_mean": 0.020105791743844748, "clip_ratio/high_max": 0.020105791743844748, "clip_ratio/region_mean": 0.051310707814991474, "reward_total_mean": 0.6803100109100342, "reward_meter_mean": 0.9477174878120422, "reward_meter_std": 0.030998004600405693, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6746211051940918, "reward_repeat_soft_std": 0.07521852850914001, "reward_judge_quality_mean": 0.2212499976158142, "reward_judge_quality_std": 0.09433034062385559, "reward_total_composite_mean": 0.6803100109100342, "reward_total_composite_std": 0.018086643889546394} {"timestamp_utc": "2026-04-13T05:02:40Z", "mode": "train", "global_step": 2660, "epoch": 0.2672024108488197, "loss": 0.0566, "grad_norm": 10.730182647705078, "learning_rate": 1.9424242424242427e-06, "num_tokens": 4946915.0, "completions/mean_length": 58.0, "completions/min_length": 49.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9492304921150208, "rewards/meter/std": 0.10523195564746857, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9733583331108093, "rewards/repeat_soft/std": 0.023218173533678055, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.8589895367622375, "rewards/total_composite/std": 0.06815318763256073, "reward": 0.8589895367622375, "reward_std": 0.06815318763256073, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09348760545253754, "sampling/sampling_logp_difference/max": 1.1439599990844727, "sampling/importance_sampling_ratio/min": 0.3185550272464752, "sampling/importance_sampling_ratio/mean": 1.0115667581558228, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.487292792648077, "clip_ratio/low_mean": 0.04043762315995991, "clip_ratio/low_min": 0.04043762315995991, "clip_ratio/high_mean": 0.025050184689462185, "clip_ratio/high_max": 0.025050184689462185, "clip_ratio/region_mean": 0.0654878078494221, "reward_total_mean": 0.8589895367622375, "reward_meter_mean": 0.9492304921150208, "reward_meter_std": 0.10523195564746857, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9733583331108093, "reward_repeat_soft_std": 0.023218173533678055, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.8589895367622375, "reward_total_composite_std": 0.06815318763256073} {"timestamp_utc": "2026-04-13T05:02:47Z", "mode": "train", "global_step": 2661, "epoch": 0.2673028628829734, "loss": -0.056, "grad_norm": 16.583858489990234, "learning_rate": 1.9393939393939395e-06, "num_tokens": 4948311.0, "completions/mean_length": 25.5, "completions/min_length": 23.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.5, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.7441034317016602, "rewards/meter/std": 0.42469486594200134, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9465834498405457, "rewards/repeat_soft/std": 0.04056889936327934, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.3341701030731201, "rewards/total_composite/mean": 0.738879919052124, "rewards/total_composite/std": 0.19359269738197327, "reward": 0.738879919052124, "reward_std": 0.19359266757965088, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10780656337738037, "sampling/sampling_logp_difference/max": 1.4183006286621094, "sampling/importance_sampling_ratio/min": 0.24212513864040375, "sampling/importance_sampling_ratio/mean": 1.0444602966308594, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5859791152179241, "clip_ratio/low_mean": 0.04732789844274521, "clip_ratio/low_min": 0.04732789844274521, "clip_ratio/high_mean": 0.046125976368784904, "clip_ratio/high_max": 0.046125976368784904, "clip_ratio/region_mean": 0.09345387481153011, "reward_total_mean": 0.738879919052124, "reward_meter_mean": 0.7441034317016602, "reward_meter_std": 0.42469486594200134, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9465834498405457, "reward_repeat_soft_std": 0.04056889936327934, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.3341701030731201, "reward_total_composite_mean": 0.738879919052124, "reward_total_composite_std": 0.19359269738197327} {"timestamp_utc": "2026-04-13T05:02:54Z", "mode": "train", "global_step": 2662, "epoch": 0.2674033149171271, "loss": 0.0108, "grad_norm": 6.9485764503479, "learning_rate": 1.9363636363636363e-06, "num_tokens": 4950370.0, "completions/mean_length": 82.375, "completions/min_length": 79.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.375, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.6624081134796143, "rewards/meter/std": 0.3134140968322754, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7168153524398804, "rewards/repeat_soft/std": 0.08210647106170654, "rewards/judge_quality/mean": 0.2837499976158142, "rewards/judge_quality/std": 0.09984809160232544, "rewards/total_composite/mean": 0.6048902273178101, "rewards/total_composite/std": 0.13486242294311523, "reward": 0.6048902273178101, "reward_std": 0.13486240804195404, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.052452266216278076, "sampling/sampling_logp_difference/max": 1.458709716796875, "sampling/importance_sampling_ratio/min": 0.23253612220287323, "sampling/importance_sampling_ratio/mean": 1.0023720264434814, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2477402240037918, "clip_ratio/low_mean": 0.018233875860460103, "clip_ratio/low_min": 0.018233875860460103, "clip_ratio/high_mean": 0.023980635683983564, "clip_ratio/high_max": 0.023980635683983564, "clip_ratio/region_mean": 0.04221451154444367, "reward_total_mean": 0.6048902273178101, "reward_meter_mean": 0.6624081134796143, "reward_meter_std": 0.3134140968322754, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7168153524398804, "reward_repeat_soft_std": 0.08210647106170654, "reward_judge_quality_mean": 0.2837499976158142, "reward_judge_quality_std": 0.09984809160232544, "reward_total_composite_mean": 0.6048902273178101, "reward_total_composite_std": 0.13486242294311523} {"timestamp_utc": "2026-04-13T05:03:02Z", "mode": "train", "global_step": 2663, "epoch": 0.26750376695128075, "loss": 0.0157, "grad_norm": 4.969394207000732, "learning_rate": 1.9333333333333336e-06, "num_tokens": 4953572.0, "completions/mean_length": 178.25, "completions/min_length": 159.0, "completions/max_length": 185.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 178.25, "completions/min_terminated_length": 159.0, "completions/max_terminated_length": 185.0, "rewards/meter/mean": 0.8917770385742188, "rewards/meter/std": 0.11378169059753418, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6924915313720703, "rewards/repeat_soft/std": 0.04963938146829605, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.23898595571517944, "rewards/total_composite/mean": 0.7322988510131836, "rewards/total_composite/std": 0.07445533573627472, "reward": 0.7322988510131836, "reward_std": 0.07445533573627472, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06547671556472778, "sampling/sampling_logp_difference/max": 2.0145087242126465, "sampling/importance_sampling_ratio/min": 0.13338592648506165, "sampling/importance_sampling_ratio/mean": 1.0045818090438843, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.30630757845938206, "clip_ratio/low_mean": 0.03289557620882988, "clip_ratio/low_min": 0.03289557620882988, "clip_ratio/high_mean": 0.02062967699021101, "clip_ratio/high_max": 0.02062967699021101, "clip_ratio/region_mean": 0.05352525319904089, "reward_total_mean": 0.7322988510131836, "reward_meter_mean": 0.8917770385742188, "reward_meter_std": 0.11378169059753418, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6924915313720703, "reward_repeat_soft_std": 0.04963938146829605, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.23898595571517944, "reward_total_composite_mean": 0.7322988510131836, "reward_total_composite_std": 0.07445533573627472} {"timestamp_utc": "2026-04-13T05:03:13Z", "mode": "train", "global_step": 2664, "epoch": 0.26760421898543446, "loss": -0.1361, "grad_norm": 2.2853682041168213, "learning_rate": 1.9303030303030304e-06, "num_tokens": 4955233.0, "completions/mean_length": 124.625, "completions/min_length": 66.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 69.28572082519531, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.8825054168701172, "rewards/meter/std": 0.28126153349876404, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7380729913711548, "rewards/repeat_soft/std": 0.11815182119607925, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.13845446705818176, "rewards/total_composite/mean": 0.6434544920921326, "rewards/total_composite/std": 0.2947236895561218, "reward": 0.6434544920921326, "reward_std": 0.2947237193584442, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07178529351949692, "sampling/sampling_logp_difference/max": 1.2153215408325195, "sampling/importance_sampling_ratio/min": 0.2966146171092987, "sampling/importance_sampling_ratio/mean": 1.0114206075668335, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4155101254582405, "clip_ratio/low_mean": 0.006410256493836641, "clip_ratio/low_min": 0.006410256493836641, "clip_ratio/high_mean": 0.04417608375661075, "clip_ratio/high_max": 0.04417608375661075, "clip_ratio/region_mean": 0.05058634025044739, "reward_total_mean": 0.6434544920921326, "reward_meter_mean": 0.8825054168701172, "reward_meter_std": 0.28126153349876404, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7380729913711548, "reward_repeat_soft_std": 0.11815182119607925, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.13845446705818176, "reward_total_composite_mean": 0.6434544920921326, "reward_total_composite_std": 0.2947236895561218} {"timestamp_utc": "2026-04-13T05:03:19Z", "mode": "train", "global_step": 2665, "epoch": 0.26770467101958817, "loss": -0.0042, "grad_norm": 11.705362319946289, "learning_rate": 1.9272727272727273e-06, "num_tokens": 4956718.0, "completions/mean_length": 26.625, "completions/min_length": 24.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.625, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9879002571105957, "rewards/meter/std": 0.003154474776238203, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9342646598815918, "rewards/repeat_soft/std": 0.04932606965303421, "rewards/judge_quality/mean": 0.38749998807907104, "rewards/judge_quality/std": 0.11877349019050598, "rewards/total_composite/mean": 0.7139619588851929, "rewards/total_composite/std": 0.2892216742038727, "reward": 0.7139619588851929, "reward_std": 0.2892216742038727, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0932910293340683, "sampling/sampling_logp_difference/max": 0.9708313941955566, "sampling/importance_sampling_ratio/min": 0.3787679970264435, "sampling/importance_sampling_ratio/mean": 1.017390251159668, "sampling/importance_sampling_ratio/max": 1.985680103302002, "entropy": 0.5102643370628357, "clip_ratio/low_mean": 0.014423076994717121, "clip_ratio/low_min": 0.014423076994717121, "clip_ratio/high_mean": 0.08040822902694345, "clip_ratio/high_max": 0.08040822902694345, "clip_ratio/region_mean": 0.09483130602166057, "reward_total_mean": 0.7139619588851929, "reward_meter_mean": 0.9879002571105957, "reward_meter_std": 0.003154474776238203, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9342646598815918, "reward_repeat_soft_std": 0.04932606965303421, "reward_judge_quality_mean": 0.38749998807907104, "reward_judge_quality_std": 0.11877349019050598, "reward_total_composite_mean": 0.7139619588851929, "reward_total_composite_std": 0.2892216742038727} {"timestamp_utc": "2026-04-13T05:03:27Z", "mode": "train", "global_step": 2666, "epoch": 0.2678051230537418, "loss": 0.0639, "grad_norm": 12.298792839050293, "learning_rate": 1.924242424242424e-06, "num_tokens": 4958152.0, "completions/mean_length": 24.25, "completions/min_length": 21.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.25, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.95509934425354, "rewards/meter/std": 0.0827014297246933, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.955390453338623, "rewards/repeat_soft/std": 0.013909636996686459, "rewards/judge_quality/mean": 0.5012500286102295, "rewards/judge_quality/std": 0.16974246501922607, "rewards/total_composite/mean": 0.8257087469100952, "rewards/total_composite/std": 0.06777754426002502, "reward": 0.8257087469100952, "reward_std": 0.06777755916118622, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09280285984277725, "sampling/sampling_logp_difference/max": 2.036996364593506, "sampling/importance_sampling_ratio/min": 0.13041986525058746, "sampling/importance_sampling_ratio/mean": 0.996369481086731, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36546681821346283, "clip_ratio/low_mean": 0.056596989277750254, "clip_ratio/low_min": 0.056596989277750254, "clip_ratio/high_mean": 0.0314685320481658, "clip_ratio/high_max": 0.0314685320481658, "clip_ratio/region_mean": 0.08806552132591605, "reward_total_mean": 0.8257087469100952, "reward_meter_mean": 0.95509934425354, "reward_meter_std": 0.0827014297246933, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.955390453338623, "reward_repeat_soft_std": 0.013909636996686459, "reward_judge_quality_mean": 0.5012500286102295, "reward_judge_quality_std": 0.16974246501922607, "reward_total_composite_mean": 0.8257087469100952, "reward_total_composite_std": 0.06777754426002502} {"timestamp_utc": "2026-04-13T05:03:33Z", "mode": "train", "global_step": 2667, "epoch": 0.26790557508789553, "loss": 0.0032, "grad_norm": 7.73460578918457, "learning_rate": 1.9212121212121213e-06, "num_tokens": 4959850.0, "completions/mean_length": 62.25, "completions/min_length": 58.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.25, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.994044303894043, "rewards/meter/std": 0.0026158420369029045, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9759763479232788, "rewards/repeat_soft/std": 0.023093143478035927, "rewards/judge_quality/mean": 0.4987499713897705, "rewards/judge_quality/std": 0.13695022463798523, "rewards/total_composite/mean": 0.8445426225662231, "rewards/total_composite/std": 0.04296927526593208, "reward": 0.8445426225662231, "reward_std": 0.042969271540641785, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0768253356218338, "sampling/sampling_logp_difference/max": 2.0151336193084717, "sampling/importance_sampling_ratio/min": 0.13330259919166565, "sampling/importance_sampling_ratio/mean": 1.0099773406982422, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.38622424006462097, "clip_ratio/low_mean": 0.050465953070670366, "clip_ratio/low_min": 0.050465953070670366, "clip_ratio/high_mean": 0.029475153423845768, "clip_ratio/high_max": 0.029475153423845768, "clip_ratio/region_mean": 0.07994110649451613, "reward_total_mean": 0.8445426225662231, "reward_meter_mean": 0.994044303894043, "reward_meter_std": 0.0026158420369029045, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9759763479232788, "reward_repeat_soft_std": 0.023093143478035927, "reward_judge_quality_mean": 0.4987499713897705, "reward_judge_quality_std": 0.13695022463798523, "reward_total_composite_mean": 0.8445426225662231, "reward_total_composite_std": 0.04296927526593208} {"timestamp_utc": "2026-04-13T05:03:40Z", "mode": "train", "global_step": 2668, "epoch": 0.26800602712204924, "loss": 0.0261, "grad_norm": 7.015196323394775, "learning_rate": 1.918181818181818e-06, "num_tokens": 4961548.0, "completions/mean_length": 42.25, "completions/min_length": 36.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.25, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.956125020980835, "rewards/meter/std": 0.013198532164096832, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8519574999809265, "rewards/repeat_soft/std": 0.10571575164794922, "rewards/judge_quality/mean": 0.3725000023841858, "rewards/judge_quality/std": 0.11055056750774384, "rewards/total_composite/mean": 0.7772020101547241, "rewards/total_composite/std": 0.03964940458536148, "reward": 0.7772020101547241, "reward_std": 0.039649397134780884, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05720829963684082, "sampling/sampling_logp_difference/max": 1.0118308067321777, "sampling/importance_sampling_ratio/min": 0.3635527789592743, "sampling/importance_sampling_ratio/mean": 1.0090082883834839, "sampling/importance_sampling_ratio/max": 1.7384307384490967, "entropy": 0.2341706696897745, "clip_ratio/low_mean": 0.02028591581620276, "clip_ratio/low_min": 0.02028591581620276, "clip_ratio/high_mean": 0.03999190451577306, "clip_ratio/high_max": 0.03999190451577306, "clip_ratio/region_mean": 0.06027782033197582, "reward_total_mean": 0.7772020101547241, "reward_meter_mean": 0.956125020980835, "reward_meter_std": 0.013198532164096832, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8519574999809265, "reward_repeat_soft_std": 0.10571575164794922, "reward_judge_quality_mean": 0.3725000023841858, "reward_judge_quality_std": 0.11055056750774384, "reward_total_composite_mean": 0.7772020101547241, "reward_total_composite_std": 0.03964940458536148} {"timestamp_utc": "2026-04-13T05:03:47Z", "mode": "train", "global_step": 2669, "epoch": 0.2681064791562029, "loss": 0.007, "grad_norm": 5.842903137207031, "learning_rate": 1.9151515151515154e-06, "num_tokens": 4964452.0, "completions/mean_length": 168.0, "completions/min_length": 145.0, "completions/max_length": 181.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 168.0, "completions/min_terminated_length": 145.0, "completions/max_terminated_length": 181.0, "rewards/meter/mean": 0.9944597482681274, "rewards/meter/std": 0.0006998951430432498, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9027565717697144, "rewards/repeat_soft/std": 0.032219961285591125, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7935326099395752, "rewards/total_composite/std": 0.022619226947426796, "reward": 0.7935326099395752, "reward_std": 0.022619234398007393, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0697624459862709, "sampling/sampling_logp_difference/max": 1.7323131561279297, "sampling/importance_sampling_ratio/min": 0.1768748015165329, "sampling/importance_sampling_ratio/mean": 1.002517580986023, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.34209785610437393, "clip_ratio/low_mean": 0.02960956282913685, "clip_ratio/low_min": 0.02960956282913685, "clip_ratio/high_mean": 0.03692384762689471, "clip_ratio/high_max": 0.03692384762689471, "clip_ratio/region_mean": 0.06653341045603156, "reward_total_mean": 0.7935326099395752, "reward_meter_mean": 0.9944597482681274, "reward_meter_std": 0.0006998951430432498, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9027565717697144, "reward_repeat_soft_std": 0.032219961285591125, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7935326099395752, "reward_total_composite_std": 0.022619226947426796} {"timestamp_utc": "2026-04-13T05:03:54Z", "mode": "train", "global_step": 2670, "epoch": 0.2682069311903566, "loss": 0.0005, "grad_norm": 11.716650009155273, "learning_rate": 1.9121212121212123e-06, "num_tokens": 4966085.0, "completions/mean_length": 56.125, "completions/min_length": 53.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.125, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9013725519180298, "rewards/meter/std": 0.20635667443275452, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9760046601295471, "rewards/repeat_soft/std": 0.023773878812789917, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.7837181091308594, "rewards/total_composite/std": 0.09462980926036835, "reward": 0.7837181091308594, "reward_std": 0.09462979435920715, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10141946375370026, "sampling/sampling_logp_difference/max": 1.2994952201843262, "sampling/importance_sampling_ratio/min": 0.30142226815223694, "sampling/importance_sampling_ratio/mean": 1.0104979276657104, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4698883034288883, "clip_ratio/low_mean": 0.007075471803545952, "clip_ratio/low_min": 0.007075471803545952, "clip_ratio/high_mean": 0.10595204122364521, "clip_ratio/high_max": 0.10595204122364521, "clip_ratio/region_mean": 0.11302751302719116, "reward_total_mean": 0.7837181091308594, "reward_meter_mean": 0.9013725519180298, "reward_meter_std": 0.20635667443275452, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9760046601295471, "reward_repeat_soft_std": 0.023773878812789917, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.7837181091308594, "reward_total_composite_std": 0.09462980926036835} {"timestamp_utc": "2026-04-13T05:04:01Z", "mode": "train", "global_step": 2671, "epoch": 0.2683073832245103, "loss": 0.02, "grad_norm": 5.967991828918457, "learning_rate": 1.9090909090909095e-06, "num_tokens": 4968718.0, "completions/mean_length": 138.125, "completions/min_length": 132.0, "completions/max_length": 156.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 138.125, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 156.0, "rewards/meter/mean": 0.9918616414070129, "rewards/meter/std": 0.002397882519289851, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8369734287261963, "rewards/repeat_soft/std": 0.08217521756887436, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.806035041809082, "rewards/total_composite/std": 0.008205348625779152, "reward": 0.806035041809082, "reward_std": 0.0082053542137146, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08630125224590302, "sampling/sampling_logp_difference/max": 3.1066832542419434, "sampling/importance_sampling_ratio/min": 0.04474913328886032, "sampling/importance_sampling_ratio/mean": 1.0000272989273071, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37254027649760246, "clip_ratio/low_mean": 0.030206505209207535, "clip_ratio/low_min": 0.030206505209207535, "clip_ratio/high_mean": 0.03678932087495923, "clip_ratio/high_max": 0.03678932087495923, "clip_ratio/region_mean": 0.06699582608416677, "reward_total_mean": 0.806035041809082, "reward_meter_mean": 0.9918616414070129, "reward_meter_std": 0.002397882519289851, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8369734287261963, "reward_repeat_soft_std": 0.08217521756887436, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.806035041809082, "reward_total_composite_std": 0.008205348625779152} {"timestamp_utc": "2026-04-13T05:04:08Z", "mode": "train", "global_step": 2672, "epoch": 0.26840783525866396, "loss": 0.0274, "grad_norm": 8.298643112182617, "learning_rate": 1.9060606060606064e-06, "num_tokens": 4970573.0, "completions/mean_length": 51.875, "completions/min_length": 42.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.875, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.8348047733306885, "rewards/meter/std": 0.29026976227760315, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9783473014831543, "rewards/repeat_soft/std": 0.04326464235782623, "rewards/judge_quality/mean": 0.7074999809265137, "rewards/judge_quality/std": 0.2474873960018158, "rewards/total_composite/mean": 0.8357468843460083, "rewards/total_composite/std": 0.13452237844467163, "reward": 0.8357468843460083, "reward_std": 0.13452237844467163, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10079183429479599, "sampling/sampling_logp_difference/max": 1.390380859375, "sampling/importance_sampling_ratio/min": 0.24898046255111694, "sampling/importance_sampling_ratio/mean": 1.0086930990219116, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47112757712602615, "clip_ratio/low_mean": 0.08074972778558731, "clip_ratio/low_min": 0.08074972778558731, "clip_ratio/high_mean": 0.039282714016735554, "clip_ratio/high_max": 0.039282714016735554, "clip_ratio/region_mean": 0.12003244180232286, "reward_total_mean": 0.8357468843460083, "reward_meter_mean": 0.8348047733306885, "reward_meter_std": 0.29026976227760315, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9783473014831543, "reward_repeat_soft_std": 0.04326464235782623, "reward_judge_quality_mean": 0.7074999809265137, "reward_judge_quality_std": 0.2474873960018158, "reward_total_composite_mean": 0.8357468843460083, "reward_total_composite_std": 0.13452237844467163} {"timestamp_utc": "2026-04-13T05:04:15Z", "mode": "train", "global_step": 2673, "epoch": 0.26850828729281767, "loss": 0.0402, "grad_norm": 7.891326904296875, "learning_rate": 1.9030303030303032e-06, "num_tokens": 4973018.0, "completions/mean_length": 123.625, "completions/min_length": 117.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 123.625, "completions/min_terminated_length": 117.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.9812747240066528, "rewards/meter/std": 0.009127260185778141, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8615859150886536, "rewards/repeat_soft/std": 0.07037695497274399, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7973572015762329, "rewards/total_composite/std": 0.01892721839249134, "reward": 0.7973572015762329, "reward_std": 0.01892721839249134, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1022840291261673, "sampling/sampling_logp_difference/max": 5.1891679763793945, "sampling/importance_sampling_ratio/min": 0.00557664455845952, "sampling/importance_sampling_ratio/mean": 1.0033478736877441, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.41297706589102745, "clip_ratio/low_mean": 0.007633587811142206, "clip_ratio/low_min": 0.007633587811142206, "clip_ratio/high_mean": 0.09705197298899293, "clip_ratio/high_max": 0.09705197298899293, "clip_ratio/region_mean": 0.10468556080013514, "reward_total_mean": 0.7973572015762329, "reward_meter_mean": 0.9812747240066528, "reward_meter_std": 0.009127260185778141, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8615859150886536, "reward_repeat_soft_std": 0.07037695497274399, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7973572015762329, "reward_total_composite_std": 0.01892721839249134} {"timestamp_utc": "2026-04-13T05:04:21Z", "mode": "train", "global_step": 2674, "epoch": 0.2686087393269714, "loss": -0.1269, "grad_norm": 21.60603141784668, "learning_rate": 1.9000000000000002e-06, "num_tokens": 4974492.0, "completions/mean_length": 26.25, "completions/min_length": 15.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.25, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.8632686138153076, "rewards/meter/std": 0.349139928817749, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.904024600982666, "rewards/repeat_soft/std": 0.16326114535331726, "rewards/judge_quality/mean": 0.3050000071525574, "rewards/judge_quality/std": 0.10928338021039963, "rewards/total_composite/mean": 0.6897482872009277, "rewards/total_composite/std": 0.2796635329723358, "reward": 0.6897482872009277, "reward_std": 0.2796635031700134, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1390678584575653, "sampling/sampling_logp_difference/max": 4.539691925048828, "sampling/importance_sampling_ratio/min": 0.010676695965230465, "sampling/importance_sampling_ratio/mean": 0.9769673943519592, "sampling/importance_sampling_ratio/max": 1.5617263317108154, "entropy": 0.7848391197621822, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10509641235694289, "clip_ratio/high_max": 0.10509641235694289, "clip_ratio/region_mean": 0.10509641235694289, "reward_total_mean": 0.6897482872009277, "reward_meter_mean": 0.8632686138153076, "reward_meter_std": 0.349139928817749, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.904024600982666, "reward_repeat_soft_std": 0.16326114535331726, "reward_judge_quality_mean": 0.3050000071525574, "reward_judge_quality_std": 0.10928338021039963, "reward_total_composite_mean": 0.6897482872009277, "reward_total_composite_std": 0.2796635329723358} {"timestamp_utc": "2026-04-13T05:04:33Z", "mode": "train", "global_step": 2675, "epoch": 0.2687091913611251, "loss": -0.1208, "grad_norm": 1.9393366575241089, "learning_rate": 1.896969696969697e-06, "num_tokens": 4976103.0, "completions/mean_length": 101.375, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 42.71428680419922, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.7389496564865112, "rewards/meter/std": 0.34639522433280945, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9715491533279419, "rewards/repeat_soft/std": 0.04355618730187416, "rewards/judge_quality/mean": 0.3675000071525574, "rewards/judge_quality/std": 0.14508618414402008, "rewards/total_composite/mean": 0.6572631001472473, "rewards/total_composite/std": 0.27651864290237427, "reward": 0.6572631001472473, "reward_std": 0.27651864290237427, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12015140056610107, "sampling/sampling_logp_difference/max": 1.2877817153930664, "sampling/importance_sampling_ratio/min": 0.2758820950984955, "sampling/importance_sampling_ratio/mean": 1.003882884979248, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47642025351524353, "clip_ratio/low_mean": 0.010204081423580647, "clip_ratio/low_min": 0.010204081423580647, "clip_ratio/high_mean": 0.10214534867554903, "clip_ratio/high_max": 0.10214534867554903, "clip_ratio/region_mean": 0.11234943009912968, "reward_total_mean": 0.6572631001472473, "reward_meter_mean": 0.7389496564865112, "reward_meter_std": 0.34639522433280945, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9715491533279419, "reward_repeat_soft_std": 0.04355618730187416, "reward_judge_quality_mean": 0.3675000071525574, "reward_judge_quality_std": 0.14508618414402008, "reward_total_composite_mean": 0.6572631001472473, "reward_total_composite_std": 0.27651864290237427} {"timestamp_utc": "2026-04-13T05:04:40Z", "mode": "train", "global_step": 2676, "epoch": 0.26880964339527874, "loss": 0.0964, "grad_norm": 8.380792617797852, "learning_rate": 1.8939393939393941e-06, "num_tokens": 4978202.0, "completions/mean_length": 91.375, "completions/min_length": 72.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.375, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.9771193265914917, "rewards/meter/std": 0.012067504227161407, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.95140141248703, "rewards/repeat_soft/std": 0.05828886479139328, "rewards/judge_quality/mean": 0.7699999809265137, "rewards/judge_quality/std": 0.22677870094776154, "rewards/total_composite/mean": 0.9158438444137573, "rewards/total_composite/std": 0.0735054463148117, "reward": 0.9158438444137573, "reward_std": 0.07350543886423111, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08898007869720459, "sampling/sampling_logp_difference/max": 1.156761646270752, "sampling/importance_sampling_ratio/min": 0.314503014087677, "sampling/importance_sampling_ratio/mean": 0.9926742911338806, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.42907996103167534, "clip_ratio/low_mean": 0.02504357835277915, "clip_ratio/low_min": 0.02504357835277915, "clip_ratio/high_mean": 0.046534314285963774, "clip_ratio/high_max": 0.046534314285963774, "clip_ratio/region_mean": 0.07157789263874292, "reward_total_mean": 0.9158438444137573, "reward_meter_mean": 0.9771193265914917, "reward_meter_std": 0.012067504227161407, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.95140141248703, "reward_repeat_soft_std": 0.05828886479139328, "reward_judge_quality_mean": 0.7699999809265137, "reward_judge_quality_std": 0.22677870094776154, "reward_total_composite_mean": 0.9158438444137573, "reward_total_composite_std": 0.0735054463148117} {"timestamp_utc": "2026-04-13T05:04:52Z", "mode": "train", "global_step": 2677, "epoch": 0.26891009542943245, "loss": -0.1498, "grad_norm": 3.1862282752990723, "learning_rate": 1.890909090909091e-06, "num_tokens": 4980214.0, "completions/mean_length": 137.5, "completions/min_length": 77.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 84.0, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.6809972524642944, "rewards/meter/std": 0.44325193762779236, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9718180894851685, "rewards/repeat_soft/std": 0.024899199604988098, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.18845234811306, "rewards/total_composite/mean": 0.6375055313110352, "rewards/total_composite/std": 0.3106050491333008, "reward": 0.6375055313110352, "reward_std": 0.3106050491333008, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11608512699604034, "sampling/sampling_logp_difference/max": 1.7483460903167725, "sampling/importance_sampling_ratio/min": 0.174061581492424, "sampling/importance_sampling_ratio/mean": 1.001011848449707, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5256140902638435, "clip_ratio/low_mean": 0.027867965400218964, "clip_ratio/low_min": 0.027867965400218964, "clip_ratio/high_mean": 0.07083258032798767, "clip_ratio/high_max": 0.07083258032798767, "clip_ratio/region_mean": 0.09870054572820663, "reward_total_mean": 0.6375055313110352, "reward_meter_mean": 0.6809972524642944, "reward_meter_std": 0.44325193762779236, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9718180894851685, "reward_repeat_soft_std": 0.024899199604988098, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.18845234811306, "reward_total_composite_mean": 0.6375055313110352, "reward_total_composite_std": 0.3106050491333008} {"timestamp_utc": "2026-04-13T05:04:59Z", "mode": "train", "global_step": 2678, "epoch": 0.26901054746358616, "loss": 0.051, "grad_norm": 6.472433567047119, "learning_rate": 1.887878787878788e-06, "num_tokens": 4982311.0, "completions/mean_length": 72.125, "completions/min_length": 65.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.125, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9858999252319336, "rewards/meter/std": 0.013888698071241379, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7447289228439331, "rewards/repeat_soft/std": 0.0960005521774292, "rewards/judge_quality/mean": 0.5724999904632568, "rewards/judge_quality/std": 0.21618445217609406, "rewards/total_composite/mean": 0.8398778438568115, "rewards/total_composite/std": 0.07423289865255356, "reward": 0.8398778438568115, "reward_std": 0.07423289120197296, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0579414889216423, "sampling/sampling_logp_difference/max": 1.4698729515075684, "sampling/importance_sampling_ratio/min": 0.22995470464229584, "sampling/importance_sampling_ratio/mean": 1.0018054246902466, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2561699356883764, "clip_ratio/low_mean": 0.026762258261442184, "clip_ratio/low_min": 0.026762258261442184, "clip_ratio/high_mean": 0.014738876488991082, "clip_ratio/high_max": 0.014738876488991082, "clip_ratio/region_mean": 0.041501134750433266, "reward_total_mean": 0.8398778438568115, "reward_meter_mean": 0.9858999252319336, "reward_meter_std": 0.013888698071241379, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7447289228439331, "reward_repeat_soft_std": 0.0960005521774292, "reward_judge_quality_mean": 0.5724999904632568, "reward_judge_quality_std": 0.21618445217609406, "reward_total_composite_mean": 0.8398778438568115, "reward_total_composite_std": 0.07423289865255356} {"timestamp_utc": "2026-04-13T05:05:06Z", "mode": "train", "global_step": 2679, "epoch": 0.2691109994977398, "loss": -0.0733, "grad_norm": 8.582549095153809, "learning_rate": 1.884848484848485e-06, "num_tokens": 4984727.0, "completions/mean_length": 122.0, "completions/min_length": 100.0, "completions/max_length": 147.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.0, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 147.0, "rewards/meter/mean": 0.9845731854438782, "rewards/meter/std": 0.02586917206645012, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9443914890289307, "rewards/repeat_soft/std": 0.03931904956698418, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7819970846176147, "rewards/total_composite/std": 0.03917906805872917, "reward": 0.7819970846176147, "reward_std": 0.039179060608148575, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10008721053600311, "sampling/sampling_logp_difference/max": 1.5365495681762695, "sampling/importance_sampling_ratio/min": 0.21512208878993988, "sampling/importance_sampling_ratio/mean": 1.017523169517517, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5311285927891731, "clip_ratio/low_mean": 0.0330555560067296, "clip_ratio/low_min": 0.0330555560067296, "clip_ratio/high_mean": 0.06352774146944284, "clip_ratio/high_max": 0.06352774146944284, "clip_ratio/region_mean": 0.09658329747617245, "reward_total_mean": 0.7819970846176147, "reward_meter_mean": 0.9845731854438782, "reward_meter_std": 0.02586917206645012, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9443914890289307, "reward_repeat_soft_std": 0.03931904956698418, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7819970846176147, "reward_total_composite_std": 0.03917906805872917} {"timestamp_utc": "2026-04-13T05:05:17Z", "mode": "train", "global_step": 2680, "epoch": 0.2692114515318935, "loss": -0.1248, "grad_norm": 3.2815709114074707, "learning_rate": 1.8818181818181819e-06, "num_tokens": 4986511.0, "completions/mean_length": 114.0, "completions/min_length": 50.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 57.142860412597656, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.764977216720581, "rewards/meter/std": 0.3578801155090332, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9777586460113525, "rewards/repeat_soft/std": 0.0372580848634243, "rewards/judge_quality/mean": 0.44749999046325684, "rewards/judge_quality/std": 0.23407875001430511, "rewards/total_composite/mean": 0.6466150283813477, "rewards/total_composite/std": 0.31428489089012146, "reward": 0.6466150283813477, "reward_std": 0.3142848610877991, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10401416569948196, "sampling/sampling_logp_difference/max": 1.7049689292907715, "sampling/importance_sampling_ratio/min": 0.18177802860736847, "sampling/importance_sampling_ratio/mean": 1.006047010421753, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4452463537454605, "clip_ratio/low_mean": 0.03675032593309879, "clip_ratio/low_min": 0.03675032593309879, "clip_ratio/high_mean": 0.06489021144807339, "clip_ratio/high_max": 0.06489021144807339, "clip_ratio/region_mean": 0.10164053738117218, "reward_total_mean": 0.6466150283813477, "reward_meter_mean": 0.764977216720581, "reward_meter_std": 0.3578801155090332, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9777586460113525, "reward_repeat_soft_std": 0.0372580848634243, "reward_judge_quality_mean": 0.44749999046325684, "reward_judge_quality_std": 0.23407875001430511, "reward_total_composite_mean": 0.6466150283813477, "reward_total_composite_std": 0.31428489089012146} {"timestamp_utc": "2026-04-13T05:05:24Z", "mode": "train", "global_step": 2681, "epoch": 0.26931190356604723, "loss": 0.0145, "grad_norm": 10.461894035339355, "learning_rate": 1.878787878787879e-06, "num_tokens": 4988781.0, "completions/mean_length": 96.75, "completions/min_length": 76.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.75, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.695976734161377, "rewards/meter/std": 0.207138791680336, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.970555305480957, "rewards/repeat_soft/std": 0.013757985085248947, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.6824951171875, "rewards/total_composite/std": 0.10455881059169769, "reward": 0.6824951171875, "reward_std": 0.1045588031411171, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12592904269695282, "sampling/sampling_logp_difference/max": 3.6885204315185547, "sampling/importance_sampling_ratio/min": 0.025008978322148323, "sampling/importance_sampling_ratio/mean": 0.9990739226341248, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5199529454112053, "clip_ratio/low_mean": 0.07726526819169521, "clip_ratio/low_min": 0.07726526819169521, "clip_ratio/high_mean": 0.05212360713630915, "clip_ratio/high_max": 0.05212360713630915, "clip_ratio/region_mean": 0.12938887532800436, "reward_total_mean": 0.6824951171875, "reward_meter_mean": 0.695976734161377, "reward_meter_std": 0.207138791680336, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.970555305480957, "reward_repeat_soft_std": 0.013757985085248947, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.6824951171875, "reward_total_composite_std": 0.10455881059169769} {"timestamp_utc": "2026-04-13T05:05:31Z", "mode": "train", "global_step": 2682, "epoch": 0.2694123556002009, "loss": -0.013, "grad_norm": 10.07761287689209, "learning_rate": 1.8757575757575757e-06, "num_tokens": 4990333.0, "completions/mean_length": 40.0, "completions/min_length": 34.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.9751326441764832, "rewards/meter/std": 0.026763668283820152, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9986742734909058, "rewards/repeat_soft/std": 0.003097930923104286, "rewards/judge_quality/mean": 0.4024999737739563, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.8094271421432495, "rewards/total_composite/std": 0.030374648049473763, "reward": 0.8094271421432495, "reward_std": 0.03037465177476406, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10975195467472076, "sampling/sampling_logp_difference/max": 1.4919638633728027, "sampling/importance_sampling_ratio/min": 0.22493049502372742, "sampling/importance_sampling_ratio/mean": 1.0157960653305054, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.554316446185112, "clip_ratio/low_mean": 0.013513513840734959, "clip_ratio/low_min": 0.013513513840734959, "clip_ratio/high_mean": 0.09063198044896126, "clip_ratio/high_max": 0.09063198044896126, "clip_ratio/region_mean": 0.10414549428969622, "reward_total_mean": 0.8094271421432495, "reward_meter_mean": 0.9751326441764832, "reward_meter_std": 0.026763668283820152, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9986742734909058, "reward_repeat_soft_std": 0.003097930923104286, "reward_judge_quality_mean": 0.4024999737739563, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.8094271421432495, "reward_total_composite_std": 0.030374648049473763} {"timestamp_utc": "2026-04-13T05:05:39Z", "mode": "train", "global_step": 2683, "epoch": 0.2695128076343546, "loss": 0.004, "grad_norm": 3.746882915496826, "learning_rate": 1.872727272727273e-06, "num_tokens": 4993495.0, "completions/mean_length": 200.25, "completions/min_length": 176.0, "completions/max_length": 211.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 200.25, "completions/min_terminated_length": 176.0, "completions/max_terminated_length": 211.0, "rewards/meter/mean": 0.9926101565361023, "rewards/meter/std": 0.004904257133603096, "rewards/count_adherence/mean": 0.9166666269302368, "rewards/count_adherence/std": 0.0890870913863182, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7492832541465759, "rewards/repeat_soft/std": 0.08927758038043976, "rewards/judge_quality/mean": 0.30124998092651367, "rewards/judge_quality/std": 0.10398317128419876, "rewards/total_composite/mean": 0.7494778633117676, "rewards/total_composite/std": 0.03235717490315437, "reward": 0.7494778633117676, "reward_std": 0.03235718607902527, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0662444606423378, "sampling/sampling_logp_difference/max": 5.882608413696289, "sampling/importance_sampling_ratio/min": 0.002787505043670535, "sampling/importance_sampling_ratio/mean": 1.0070668458938599, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.293495774269104, "clip_ratio/low_mean": 0.018286177655681968, "clip_ratio/low_min": 0.018286177655681968, "clip_ratio/high_mean": 0.037250333931297064, "clip_ratio/high_max": 0.037250333931297064, "clip_ratio/region_mean": 0.05553651158697903, "reward_total_mean": 0.7494778633117676, "reward_meter_mean": 0.9926101565361023, "reward_meter_std": 0.004904257133603096, "reward_count_adherence_mean": 0.9166666269302368, "reward_count_adherence_std": 0.0890870913863182, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7492832541465759, "reward_repeat_soft_std": 0.08927758038043976, "reward_judge_quality_mean": 0.30124998092651367, "reward_judge_quality_std": 0.10398317128419876, "reward_total_composite_mean": 0.7494778633117676, "reward_total_composite_std": 0.03235717490315437} {"timestamp_utc": "2026-04-13T05:05:45Z", "mode": "train", "global_step": 2684, "epoch": 0.2696132596685083, "loss": 0.0063, "grad_norm": 9.126130104064941, "learning_rate": 1.86969696969697e-06, "num_tokens": 4995240.0, "completions/mean_length": 53.125, "completions/min_length": 45.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.125, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9351728558540344, "rewards/meter/std": 0.09813269972801208, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9576681852340698, "rewards/repeat_soft/std": 0.01597709208726883, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.8113446235656738, "rewards/total_composite/std": 0.019086966291069984, "reward": 0.8113446235656738, "reward_std": 0.019086968153715134, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1030518040060997, "sampling/sampling_logp_difference/max": 1.149076223373413, "sampling/importance_sampling_ratio/min": 0.316929429769516, "sampling/importance_sampling_ratio/mean": 1.0124491453170776, "sampling/importance_sampling_ratio/max": 1.9778722524642944, "entropy": 0.4993631839752197, "clip_ratio/low_mean": 0.03999779000878334, "clip_ratio/low_min": 0.03999779000878334, "clip_ratio/high_mean": 0.06441212492063642, "clip_ratio/high_max": 0.06441212492063642, "clip_ratio/region_mean": 0.10440991492941976, "reward_total_mean": 0.8113446235656738, "reward_meter_mean": 0.9351728558540344, "reward_meter_std": 0.09813269972801208, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9576681852340698, "reward_repeat_soft_std": 0.01597709208726883, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.8113446235656738, "reward_total_composite_std": 0.019086966291069984} {"timestamp_utc": "2026-04-13T05:05:51Z", "mode": "train", "global_step": 2685, "epoch": 0.26971371170266195, "loss": 0.0125, "grad_norm": 9.721368789672852, "learning_rate": 1.8666666666666669e-06, "num_tokens": 4996707.0, "completions/mean_length": 24.375, "completions/min_length": 23.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.375, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.9652915000915527, "rewards/meter/std": 0.0349326990544796, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9466345906257629, "rewards/repeat_soft/std": 0.02364136464893818, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8106696605682373, "rewards/total_composite/std": 0.019154109060764313, "reward": 0.8106696605682373, "reward_std": 0.019154109060764313, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10100638121366501, "sampling/sampling_logp_difference/max": 1.4551994800567627, "sampling/importance_sampling_ratio/min": 0.23335380852222443, "sampling/importance_sampling_ratio/mean": 0.9941802620887756, "sampling/importance_sampling_ratio/max": 1.789779782295227, "entropy": 0.5254813432693481, "clip_ratio/low_mean": 0.01963141094893217, "clip_ratio/low_min": 0.01963141094893217, "clip_ratio/high_mean": 0.037155797239392996, "clip_ratio/high_max": 0.037155797239392996, "clip_ratio/region_mean": 0.05678720818832517, "reward_total_mean": 0.8106696605682373, "reward_meter_mean": 0.9652915000915527, "reward_meter_std": 0.0349326990544796, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9466345906257629, "reward_repeat_soft_std": 0.02364136464893818, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8106696605682373, "reward_total_composite_std": 0.019154109060764313} {"timestamp_utc": "2026-04-13T05:05:58Z", "mode": "train", "global_step": 2686, "epoch": 0.26981416373681566, "loss": 0.0047, "grad_norm": 10.257284164428711, "learning_rate": 1.863636363636364e-06, "num_tokens": 4998501.0, "completions/mean_length": 68.25, "completions/min_length": 63.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.25, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9851977825164795, "rewards/meter/std": 0.011320519261062145, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9296835064888, "rewards/repeat_soft/std": 0.0491153784096241, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.8059324026107788, "rewards/total_composite/std": 0.01841202937066555, "reward": 0.8059324026107788, "reward_std": 0.018412036821246147, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.099154993891716, "sampling/sampling_logp_difference/max": 2.149442672729492, "sampling/importance_sampling_ratio/min": 0.11654909700155258, "sampling/importance_sampling_ratio/mean": 0.9976138472557068, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5143088176846504, "clip_ratio/low_mean": 0.014657444320619106, "clip_ratio/low_min": 0.014657444320619106, "clip_ratio/high_mean": 0.08216295763850212, "clip_ratio/high_max": 0.08216295763850212, "clip_ratio/region_mean": 0.09682040195912123, "reward_total_mean": 0.8059324026107788, "reward_meter_mean": 0.9851977825164795, "reward_meter_std": 0.011320519261062145, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9296835064888, "reward_repeat_soft_std": 0.0491153784096241, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.8059324026107788, "reward_total_composite_std": 0.01841202937066555} {"timestamp_utc": "2026-04-13T05:06:05Z", "mode": "train", "global_step": 2687, "epoch": 0.26991461577096937, "loss": -0.0048, "grad_norm": 7.182571887969971, "learning_rate": 1.8606060606060607e-06, "num_tokens": 5000378.0, "completions/mean_length": 81.625, "completions/min_length": 76.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.625, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9867929220199585, "rewards/meter/std": 0.005253981798887253, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9479931592941284, "rewards/repeat_soft/std": 0.03143614903092384, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8148561716079712, "rewards/total_composite/std": 0.004285949282348156, "reward": 0.8148561716079712, "reward_std": 0.004285944160073996, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08524198085069656, "sampling/sampling_logp_difference/max": 1.6905698776245117, "sampling/importance_sampling_ratio/min": 0.18441438674926758, "sampling/importance_sampling_ratio/mean": 1.0185827016830444, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39813095331192017, "clip_ratio/low_mean": 0.02742625679820776, "clip_ratio/low_min": 0.02742625679820776, "clip_ratio/high_mean": 0.0536390277557075, "clip_ratio/high_max": 0.0536390277557075, "clip_ratio/region_mean": 0.08106528455391526, "reward_total_mean": 0.8148561716079712, "reward_meter_mean": 0.9867929220199585, "reward_meter_std": 0.005253981798887253, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9479931592941284, "reward_repeat_soft_std": 0.03143614903092384, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8148561716079712, "reward_total_composite_std": 0.004285949282348156} {"timestamp_utc": "2026-04-13T05:06:12Z", "mode": "train", "global_step": 2688, "epoch": 0.2700150678051231, "loss": -0.0157, "grad_norm": 6.886253833770752, "learning_rate": 1.8575757575757578e-06, "num_tokens": 5002456.0, "completions/mean_length": 96.75, "completions/min_length": 66.0, "completions/max_length": 119.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.75, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.9620835781097412, "rewards/meter/std": 0.0444866418838501, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.824682354927063, "rewards/repeat_soft/std": 0.08201666176319122, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7820308208465576, "rewards/total_composite/std": 0.019800279289484024, "reward": 0.7820308208465576, "reward_std": 0.019800279289484024, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07310675829648972, "sampling/sampling_logp_difference/max": 3.1646788120269775, "sampling/importance_sampling_ratio/min": 0.04222770407795906, "sampling/importance_sampling_ratio/mean": 0.9985700249671936, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2875418942421675, "clip_ratio/low_mean": 0.025614352663978934, "clip_ratio/low_min": 0.025614352663978934, "clip_ratio/high_mean": 0.027390277944505215, "clip_ratio/high_max": 0.027390277944505215, "clip_ratio/region_mean": 0.05300463060848415, "reward_total_mean": 0.7820308208465576, "reward_meter_mean": 0.9620835781097412, "reward_meter_std": 0.0444866418838501, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.824682354927063, "reward_repeat_soft_std": 0.08201666176319122, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7820308208465576, "reward_total_composite_std": 0.019800279289484024} {"timestamp_utc": "2026-04-13T05:06:18Z", "mode": "train", "global_step": 2689, "epoch": 0.27011551983927673, "loss": -0.0447, "grad_norm": 13.6732759475708, "learning_rate": 1.8545454545454546e-06, "num_tokens": 5003955.0, "completions/mean_length": 29.375, "completions/min_length": 25.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.375, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9841536283493042, "rewards/meter/std": 0.004720176570117474, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9412878751754761, "rewards/repeat_soft/std": 0.02516329102218151, "rewards/judge_quality/mean": 0.5637500286102295, "rewards/judge_quality/std": 0.22012579441070557, "rewards/total_composite/mean": 0.8561228513717651, "rewards/total_composite/std": 0.06643134355545044, "reward": 0.8561228513717651, "reward_std": 0.06643134355545044, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08945807814598083, "sampling/sampling_logp_difference/max": 1.0261011123657227, "sampling/importance_sampling_ratio/min": 0.3584016263484955, "sampling/importance_sampling_ratio/mean": 1.0093573331832886, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.46428582817316055, "clip_ratio/low_mean": 0.03669709339737892, "clip_ratio/low_min": 0.03669709339737892, "clip_ratio/high_mean": 0.023347701877355576, "clip_ratio/high_max": 0.023347701877355576, "clip_ratio/region_mean": 0.0600447952747345, "reward_total_mean": 0.8561228513717651, "reward_meter_mean": 0.9841536283493042, "reward_meter_std": 0.004720176570117474, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9412878751754761, "reward_repeat_soft_std": 0.02516329102218151, "reward_judge_quality_mean": 0.5637500286102295, "reward_judge_quality_std": 0.22012579441070557, "reward_total_composite_mean": 0.8561228513717651, "reward_total_composite_std": 0.06643134355545044} {"timestamp_utc": "2026-04-13T05:06:24Z", "mode": "train", "global_step": 2690, "epoch": 0.27021597187343044, "loss": -0.0102, "grad_norm": 11.239594459533691, "learning_rate": 1.8515151515151517e-06, "num_tokens": 5005582.0, "completions/mean_length": 54.375, "completions/min_length": 48.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.375, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9522559642791748, "rewards/meter/std": 0.06499030441045761, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9856773614883423, "rewards/repeat_soft/std": 0.014839740470051765, "rewards/judge_quality/mean": 0.41749998927116394, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.802332878112793, "rewards/total_composite/std": 0.030593860894441605, "reward": 0.802332878112793, "reward_std": 0.030593853443861008, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0896550640463829, "sampling/sampling_logp_difference/max": 1.9525327682495117, "sampling/importance_sampling_ratio/min": 0.3143620789051056, "sampling/importance_sampling_ratio/mean": 1.0160092115402222, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5036537609994411, "clip_ratio/low_mean": 0.040285118855535984, "clip_ratio/low_min": 0.040285118855535984, "clip_ratio/high_mean": 0.03522730083204806, "clip_ratio/high_max": 0.03522730083204806, "clip_ratio/region_mean": 0.07551241968758404, "reward_total_mean": 0.802332878112793, "reward_meter_mean": 0.9522559642791748, "reward_meter_std": 0.06499030441045761, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9856773614883423, "reward_repeat_soft_std": 0.014839740470051765, "reward_judge_quality_mean": 0.41749998927116394, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.802332878112793, "reward_total_composite_std": 0.030593860894441605} {"timestamp_utc": "2026-04-13T05:06:31Z", "mode": "train", "global_step": 2691, "epoch": 0.27031642390758415, "loss": 0.0232, "grad_norm": 13.80778980255127, "learning_rate": 1.8484848484848487e-06, "num_tokens": 5007267.0, "completions/mean_length": 42.625, "completions/min_length": 40.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.625, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.9524820446968079, "rewards/meter/std": 0.014733112417161465, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8731520771980286, "rewards/repeat_soft/std": 0.1130414754152298, "rewards/judge_quality/mean": 0.3174999952316284, "rewards/judge_quality/std": 0.0936177596449852, "rewards/total_composite/mean": 0.7611821293830872, "rewards/total_composite/std": 0.033802613615989685, "reward": 0.7611821293830872, "reward_std": 0.03380262106657028, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09207554161548615, "sampling/sampling_logp_difference/max": 1.2753753662109375, "sampling/importance_sampling_ratio/min": 0.27932611107826233, "sampling/importance_sampling_ratio/mean": 0.9940013885498047, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.38616666570305824, "clip_ratio/low_mean": 0.0302290974650532, "clip_ratio/low_min": 0.0302290974650532, "clip_ratio/high_mean": 0.04005168005824089, "clip_ratio/high_max": 0.04005168005824089, "clip_ratio/region_mean": 0.07028077752329409, "reward_total_mean": 0.7611821293830872, "reward_meter_mean": 0.9524820446968079, "reward_meter_std": 0.014733112417161465, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8731520771980286, "reward_repeat_soft_std": 0.1130414754152298, "reward_judge_quality_mean": 0.3174999952316284, "reward_judge_quality_std": 0.0936177596449852, "reward_total_composite_mean": 0.7611821293830872, "reward_total_composite_std": 0.033802613615989685} {"timestamp_utc": "2026-04-13T05:06:36Z", "mode": "train", "global_step": 2692, "epoch": 0.2704168759417378, "loss": -0.0099, "grad_norm": 15.468896865844727, "learning_rate": 1.8454545454545455e-06, "num_tokens": 5008635.0, "completions/mean_length": 28.0, "completions/min_length": 25.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.0, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.9902231097221375, "rewards/meter/std": 0.003666984150186181, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9556125402450562, "rewards/repeat_soft/std": 0.019480695948004723, "rewards/judge_quality/mean": 0.6850000023841858, "rewards/judge_quality/std": 0.2512255907058716, "rewards/total_composite/mean": 0.896661639213562, "rewards/total_composite/std": 0.0748976394534111, "reward": 0.896661639213562, "reward_std": 0.0748976469039917, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09627551585435867, "sampling/sampling_logp_difference/max": 1.1433619260787964, "sampling/importance_sampling_ratio/min": 0.31874561309814453, "sampling/importance_sampling_ratio/mean": 0.9947897791862488, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5997913554310799, "clip_ratio/low_mean": 0.05619658203795552, "clip_ratio/low_min": 0.05619658203795552, "clip_ratio/high_mean": 0.030172414612025023, "clip_ratio/high_max": 0.030172414612025023, "clip_ratio/region_mean": 0.08636899664998055, "reward_total_mean": 0.896661639213562, "reward_meter_mean": 0.9902231097221375, "reward_meter_std": 0.003666984150186181, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9556125402450562, "reward_repeat_soft_std": 0.019480695948004723, "reward_judge_quality_mean": 0.6850000023841858, "reward_judge_quality_std": 0.2512255907058716, "reward_total_composite_mean": 0.896661639213562, "reward_total_composite_std": 0.0748976394534111} {"timestamp_utc": "2026-04-13T05:06:43Z", "mode": "train", "global_step": 2693, "epoch": 0.2705173279758915, "loss": 0.0313, "grad_norm": 28.245052337646484, "learning_rate": 1.8424242424242426e-06, "num_tokens": 5010574.0, "completions/mean_length": 83.375, "completions/min_length": 75.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.375, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9611292481422424, "rewards/meter/std": 0.06242553889751434, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9289465546607971, "rewards/repeat_soft/std": 0.037210218608379364, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404937028885, "rewards/total_composite/mean": 0.8422778248786926, "rewards/total_composite/std": 0.07981374114751816, "reward": 0.8422778248786926, "reward_std": 0.07981374859809875, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08654476702213287, "sampling/sampling_logp_difference/max": 1.3846731185913086, "sampling/importance_sampling_ratio/min": 0.2504056394100189, "sampling/importance_sampling_ratio/mean": 1.0013148784637451, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3644774854183197, "clip_ratio/low_mean": 0.044497463502921164, "clip_ratio/low_min": 0.044497463502921164, "clip_ratio/high_mean": 0.02059108577668667, "clip_ratio/high_max": 0.02059108577668667, "clip_ratio/region_mean": 0.06508854927960783, "reward_total_mean": 0.8422778248786926, "reward_meter_mean": 0.9611292481422424, "reward_meter_std": 0.06242553889751434, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9289465546607971, "reward_repeat_soft_std": 0.037210218608379364, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404937028885, "reward_total_composite_mean": 0.8422778248786926, "reward_total_composite_std": 0.07981374114751816} {"timestamp_utc": "2026-04-13T05:06:54Z", "mode": "train", "global_step": 2694, "epoch": 0.2706177800100452, "loss": -0.0961, "grad_norm": 1.7794705629348755, "learning_rate": 1.8393939393939394e-06, "num_tokens": 5012098.0, "completions/mean_length": 89.5, "completions/min_length": 27.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 29.142858505249023, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.8753827810287476, "rewards/meter/std": 0.3284662663936615, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9530420899391174, "rewards/repeat_soft/std": 0.014448494650423527, "rewards/judge_quality/mean": 0.3174999952316284, "rewards/judge_quality/std": 0.14210157096385956, "rewards/total_composite/mean": 0.6983010172843933, "rewards/total_composite/std": 0.283484548330307, "reward": 0.6983010172843933, "reward_std": 0.2834845185279846, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1316719949245453, "sampling/sampling_logp_difference/max": 1.2845959663391113, "sampling/importance_sampling_ratio/min": 0.27676236629486084, "sampling/importance_sampling_ratio/mean": 0.9846928715705872, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5496631488204002, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.1243564416654408, "clip_ratio/high_max": 0.1243564416654408, "clip_ratio/region_mean": 0.1243564416654408, "reward_total_mean": 0.6983010172843933, "reward_meter_mean": 0.8753827810287476, "reward_meter_std": 0.3284662663936615, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9530420899391174, "reward_repeat_soft_std": 0.014448494650423527, "reward_judge_quality_mean": 0.3174999952316284, "reward_judge_quality_std": 0.14210157096385956, "reward_total_composite_mean": 0.6983010172843933, "reward_total_composite_std": 0.283484548330307} {"timestamp_utc": "2026-04-13T05:07:01Z", "mode": "train", "global_step": 2695, "epoch": 0.27071823204419887, "loss": 0.0386, "grad_norm": 6.978457450866699, "learning_rate": 1.8363636363636365e-06, "num_tokens": 5014223.0, "completions/mean_length": 82.625, "completions/min_length": 72.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.625, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.88149094581604, "rewards/meter/std": 0.23108041286468506, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8600844144821167, "rewards/repeat_soft/std": 0.049750953912734985, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.20119288563728333, "rewards/total_composite/mean": 0.7772418260574341, "rewards/total_composite/std": 0.1076279729604721, "reward": 0.7772418260574341, "reward_std": 0.10762795060873032, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06760873645544052, "sampling/sampling_logp_difference/max": 1.4155304431915283, "sampling/importance_sampling_ratio/min": 0.24279679358005524, "sampling/importance_sampling_ratio/mean": 1.000728726387024, "sampling/importance_sampling_ratio/max": 1.9908918142318726, "entropy": 0.2511104252189398, "clip_ratio/low_mean": 0.011651632376015186, "clip_ratio/low_min": 0.011651632376015186, "clip_ratio/high_mean": 0.04707015352323651, "clip_ratio/high_max": 0.04707015352323651, "clip_ratio/region_mean": 0.0587217858992517, "reward_total_mean": 0.7772418260574341, "reward_meter_mean": 0.88149094581604, "reward_meter_std": 0.23108041286468506, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8600844144821167, "reward_repeat_soft_std": 0.049750953912734985, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.20119288563728333, "reward_total_composite_mean": 0.7772418260574341, "reward_total_composite_std": 0.1076279729604721} {"timestamp_utc": "2026-04-13T05:07:08Z", "mode": "train", "global_step": 2696, "epoch": 0.2708186840783526, "loss": -0.0058, "grad_norm": 5.44091272354126, "learning_rate": 1.8333333333333333e-06, "num_tokens": 5016689.0, "completions/mean_length": 117.25, "completions/min_length": 105.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.25, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9827884435653687, "rewards/meter/std": 0.005673691164702177, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8892772197723389, "rewards/repeat_soft/std": 0.06647484004497528, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7961200475692749, "rewards/total_composite/std": 0.03618894889950752, "reward": 0.7961200475692749, "reward_std": 0.03618892282247543, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07299377769231796, "sampling/sampling_logp_difference/max": 2.1799771785736084, "sampling/importance_sampling_ratio/min": 0.11304411292076111, "sampling/importance_sampling_ratio/mean": 1.0011835098266602, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.38337112963199615, "clip_ratio/low_mean": 0.004629629664123058, "clip_ratio/low_min": 0.004629629664123058, "clip_ratio/high_mean": 0.06764733511954546, "clip_ratio/high_max": 0.06764733511954546, "clip_ratio/region_mean": 0.07227696478366852, "reward_total_mean": 0.7961200475692749, "reward_meter_mean": 0.9827884435653687, "reward_meter_std": 0.005673691164702177, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8892772197723389, "reward_repeat_soft_std": 0.06647484004497528, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7961200475692749, "reward_total_composite_std": 0.03618894889950752} {"timestamp_utc": "2026-04-13T05:07:15Z", "mode": "train", "global_step": 2697, "epoch": 0.2709191361125063, "loss": 0.0438, "grad_norm": 9.950935363769531, "learning_rate": 1.8303030303030305e-06, "num_tokens": 5018555.0, "completions/mean_length": 64.25, "completions/min_length": 55.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.25, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.992371141910553, "rewards/meter/std": 0.006842368748039007, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.975307822227478, "rewards/repeat_soft/std": 0.016373485326766968, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.820097804069519, "rewards/total_composite/std": 0.0025486410595476627, "reward": 0.820097804069519, "reward_std": 0.0025486398953944445, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10736497491598129, "sampling/sampling_logp_difference/max": 1.5475825071334839, "sampling/importance_sampling_ratio/min": 0.21276170015335083, "sampling/importance_sampling_ratio/mean": 1.007165789604187, "sampling/importance_sampling_ratio/max": 1.7459453344345093, "entropy": 0.6167127266526222, "clip_ratio/low_mean": 0.02602163515985012, "clip_ratio/low_min": 0.02602163515985012, "clip_ratio/high_mean": 0.0713588004000485, "clip_ratio/high_max": 0.0713588004000485, "clip_ratio/region_mean": 0.09738043555989861, "reward_total_mean": 0.820097804069519, "reward_meter_mean": 0.992371141910553, "reward_meter_std": 0.006842368748039007, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.975307822227478, "reward_repeat_soft_std": 0.016373485326766968, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.820097804069519, "reward_total_composite_std": 0.0025486410595476627} {"timestamp_utc": "2026-04-13T05:07:22Z", "mode": "train", "global_step": 2698, "epoch": 0.27101958814666, "loss": 0.093, "grad_norm": 13.035334587097168, "learning_rate": 1.8272727272727276e-06, "num_tokens": 5020524.0, "completions/mean_length": 71.125, "completions/min_length": 64.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.125, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.8583636283874512, "rewards/meter/std": 0.3267802894115448, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9662941098213196, "rewards/repeat_soft/std": 0.02004249580204487, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.0975411981344223, "rewards/total_composite/mean": 0.7405180335044861, "rewards/total_composite/std": 0.20112599432468414, "reward": 0.7405180335044861, "reward_std": 0.20112602412700653, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13423842191696167, "sampling/sampling_logp_difference/max": 1.7369270324707031, "sampling/importance_sampling_ratio/min": 0.17606060206890106, "sampling/importance_sampling_ratio/mean": 1.0126595497131348, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8954334445297718, "clip_ratio/low_mean": 0.01718750037252903, "clip_ratio/low_min": 0.01718750037252903, "clip_ratio/high_mean": 0.0971287009306252, "clip_ratio/high_max": 0.0971287009306252, "clip_ratio/region_mean": 0.11431620130315423, "reward_total_mean": 0.7405180335044861, "reward_meter_mean": 0.8583636283874512, "reward_meter_std": 0.3267802894115448, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9662941098213196, "reward_repeat_soft_std": 0.02004249580204487, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.0975411981344223, "reward_total_composite_mean": 0.7405180335044861, "reward_total_composite_std": 0.20112599432468414} {"timestamp_utc": "2026-04-13T05:07:30Z", "mode": "train", "global_step": 2699, "epoch": 0.27112004018081365, "loss": 0.0296, "grad_norm": 7.7433342933654785, "learning_rate": 1.8242424242424244e-06, "num_tokens": 5023102.0, "completions/mean_length": 139.25, "completions/min_length": 126.0, "completions/max_length": 165.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 139.25, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 165.0, "rewards/meter/mean": 0.834324836730957, "rewards/meter/std": 0.32119616866111755, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8217083215713501, "rewards/repeat_soft/std": 0.03995592147111893, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.6177235841751099, "rewards/total_composite/std": 0.2841551899909973, "reward": 0.6177235841751099, "reward_std": 0.2841551601886749, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09561583399772644, "sampling/sampling_logp_difference/max": 2.5832018852233887, "sampling/importance_sampling_ratio/min": 0.07553177326917648, "sampling/importance_sampling_ratio/mean": 1.0058523416519165, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.29814800806343555, "clip_ratio/low_mean": 0.022112861275672913, "clip_ratio/low_min": 0.022112861275672913, "clip_ratio/high_mean": 0.061324607115238905, "clip_ratio/high_max": 0.061324607115238905, "clip_ratio/region_mean": 0.08343746839091182, "reward_total_mean": 0.6177235841751099, "reward_meter_mean": 0.834324836730957, "reward_meter_std": 0.32119616866111755, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8217083215713501, "reward_repeat_soft_std": 0.03995592147111893, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.6177235841751099, "reward_total_composite_std": 0.2841551899909973} {"timestamp_utc": "2026-04-13T05:07:36Z", "mode": "train", "global_step": 2700, "epoch": 0.27122049221496736, "loss": -0.0018, "grad_norm": 16.257123947143555, "learning_rate": 1.8212121212121215e-06, "num_tokens": 5024490.0, "completions/mean_length": 31.5, "completions/min_length": 27.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.5, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.8547145128250122, "rewards/meter/std": 0.23952700197696686, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9565702676773071, "rewards/repeat_soft/std": 0.015635881572961807, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.7795284986495972, "rewards/total_composite/std": 0.1300581693649292, "reward": 0.7795284986495972, "reward_std": 0.1300581693649292, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.125385120511055, "sampling/sampling_logp_difference/max": 1.5825624465942383, "sampling/importance_sampling_ratio/min": 0.2054479718208313, "sampling/importance_sampling_ratio/mean": 0.9999445080757141, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5581529624760151, "clip_ratio/low_mean": 0.028823177330195904, "clip_ratio/low_min": 0.028823177330195904, "clip_ratio/high_mean": 0.07764134649187326, "clip_ratio/high_max": 0.07764134649187326, "clip_ratio/region_mean": 0.10646452382206917, "reward_total_mean": 0.7795284986495972, "reward_meter_mean": 0.8547145128250122, "reward_meter_std": 0.23952700197696686, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9565702676773071, "reward_repeat_soft_std": 0.015635881572961807, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.7795284986495972, "reward_total_composite_std": 0.1300581693649292} {"timestamp_utc": "2026-04-13T05:08:32Z", "mode": "eval", "global_step": 2700, "epoch": 0.27122049221496736, "eval_loss": NaN, "eval_runtime": 56.6267, "eval_samples_per_second": 1.413, "eval_steps_per_second": 0.177, "eval_num_tokens": 5024490.0, "eval_completions/mean_length": 109.3, "eval_completions/min_length": 40.6, "eval_completions/max_length": 261.1, "eval_completions/clipped_ratio": 0.0375, "eval_completions/mean_terminated_length": 93.64464416503907, "eval_completions/min_terminated_length": 40.6, "eval_completions/max_terminated_length": 156.8, "eval_rewards/meter/mean": 0.8763339757919312, "eval_rewards/meter/std": 0.19059391326736658, "eval_rewards/count_adherence/mean": 0.9806250035762787, "eval_rewards/count_adherence/std": 0.049916839599609374, "eval_rewards/hard_gate/mean": 0.9625, "eval_rewards/hard_gate/std": 0.10606601536273956, "eval_rewards/repeat_soft/mean": 0.8964724540710449, "eval_rewards/repeat_soft/std": 0.09359675571322441, "eval_rewards/judge_quality/mean": 0.41449999809265137, "eval_rewards/judge_quality/std": 0.14052291912958026, "eval_rewards/total_composite/mean": 0.736965560913086, "eval_rewards/total_composite/std": 0.15595460385084153, "eval_reward": 0.736965560913086, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.03921934589743614, "eval_sampling/sampling_logp_difference/max": 0.8546501159667969, "eval_sampling/importance_sampling_ratio/min": 0.43354603350162507, "eval_sampling/importance_sampling_ratio/mean": 1.0093213081359864, "eval_sampling/importance_sampling_ratio/max": 1.3319107174873352, "eval_entropy": 0.4124505013227463, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.736965560913086, "eval_reward_meter_mean": 0.8763339757919312, "eval_reward_meter_std": 0.19059391326736658, "eval_reward_count_adherence_mean": 0.9806250035762787, "eval_reward_count_adherence_std": 0.049916839599609374, "eval_reward_hard_gate_mean": 0.9625, "eval_reward_hard_gate_std": 0.10606601536273956, "eval_reward_repeat_soft_mean": 0.8964724540710449, "eval_reward_repeat_soft_std": 0.09359675571322441, "eval_reward_judge_quality_mean": 0.41449999809265137, "eval_reward_judge_quality_std": 0.14052291912958026, "eval_reward_total_composite_mean": 0.736965560913086, "eval_reward_total_composite_std": 0.15595460385084153} {"timestamp_utc": "2026-04-13T05:08:42Z", "mode": "train", "global_step": 2701, "epoch": 0.27132094424912107, "loss": 0.0322, "grad_norm": 5.402583599090576, "learning_rate": 1.8181818181818183e-06, "num_tokens": 5027022.0, "completions/mean_length": 127.5, "completions/min_length": 121.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.5, "completions/min_terminated_length": 121.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.9127991199493408, "rewards/meter/std": 0.11409544944763184, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9384907484054565, "rewards/repeat_soft/std": 0.042429715394973755, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7678586840629578, "rewards/total_composite/std": 0.046263497322797775, "reward": 0.7678586840629578, "reward_std": 0.046263474971055984, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08625870198011398, "sampling/sampling_logp_difference/max": 2.227374315261841, "sampling/importance_sampling_ratio/min": 0.10781113803386688, "sampling/importance_sampling_ratio/mean": 0.9997918009757996, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43767427280545235, "clip_ratio/low_mean": 0.03658545855432749, "clip_ratio/low_min": 0.03658545855432749, "clip_ratio/high_mean": 0.02360950643196702, "clip_ratio/high_max": 0.02360950643196702, "clip_ratio/region_mean": 0.06019496498629451, "reward_total_mean": 0.7678586840629578, "reward_meter_mean": 0.9127991199493408, "reward_meter_std": 0.11409544944763184, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9384907484054565, "reward_repeat_soft_std": 0.042429715394973755, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7678586840629578, "reward_total_composite_std": 0.046263497322797775} {"timestamp_utc": "2026-04-13T05:08:49Z", "mode": "train", "global_step": 2702, "epoch": 0.2714213962832747, "loss": -0.0091, "grad_norm": 7.662420272827148, "learning_rate": 1.8151515151515153e-06, "num_tokens": 5028668.0, "completions/mean_length": 51.75, "completions/min_length": 44.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.75, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9880987405776978, "rewards/meter/std": 0.002080738777294755, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8607316017150879, "rewards/repeat_soft/std": 0.05825423076748848, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8067175149917603, "rewards/total_composite/std": 0.005744580645114183, "reward": 0.8067175149917603, "reward_std": 0.005744589492678642, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05688689276576042, "sampling/sampling_logp_difference/max": 1.136181354522705, "sampling/importance_sampling_ratio/min": 0.32104262709617615, "sampling/importance_sampling_ratio/mean": 1.0080373287200928, "sampling/importance_sampling_ratio/max": 1.5869938135147095, "entropy": 0.32387941889464855, "clip_ratio/low_mean": 0.033301768358796835, "clip_ratio/low_min": 0.033301768358796835, "clip_ratio/high_mean": 0.029944798443466425, "clip_ratio/high_max": 0.029944798443466425, "clip_ratio/region_mean": 0.06324656680226326, "reward_total_mean": 0.8067175149917603, "reward_meter_mean": 0.9880987405776978, "reward_meter_std": 0.002080738777294755, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8607316017150879, "reward_repeat_soft_std": 0.05825423076748848, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8067175149917603, "reward_total_composite_std": 0.005744580645114183} {"timestamp_utc": "2026-04-13T05:08:55Z", "mode": "train", "global_step": 2703, "epoch": 0.2715218483174284, "loss": 0.0276, "grad_norm": 8.298543930053711, "learning_rate": 1.8121212121212124e-06, "num_tokens": 5030558.0, "completions/mean_length": 61.25, "completions/min_length": 58.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.25, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9861221313476562, "rewards/meter/std": 0.009276560507714748, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9469962120056152, "rewards/repeat_soft/std": 0.02401629090309143, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8167046308517456, "rewards/total_composite/std": 0.008452902548015118, "reward": 0.8167046308517456, "reward_std": 0.008452891372144222, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07562333345413208, "sampling/sampling_logp_difference/max": 1.077336072921753, "sampling/importance_sampling_ratio/min": 0.3405013978481293, "sampling/importance_sampling_ratio/mean": 1.0206003189086914, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44021253287792206, "clip_ratio/low_mean": 0.030034761177375913, "clip_ratio/low_min": 0.030034761177375913, "clip_ratio/high_mean": 0.0287151075899601, "clip_ratio/high_max": 0.0287151075899601, "clip_ratio/region_mean": 0.05874986876733601, "reward_total_mean": 0.8167046308517456, "reward_meter_mean": 0.9861221313476562, "reward_meter_std": 0.009276560507714748, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9469962120056152, "reward_repeat_soft_std": 0.02401629090309143, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8167046308517456, "reward_total_composite_std": 0.008452902548015118} {"timestamp_utc": "2026-04-13T05:09:03Z", "mode": "train", "global_step": 2704, "epoch": 0.27162230035158214, "loss": 0.0769, "grad_norm": 5.711086273193359, "learning_rate": 1.8090909090909092e-06, "num_tokens": 5032975.0, "completions/mean_length": 128.125, "completions/min_length": 117.0, "completions/max_length": 146.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.125, "completions/min_terminated_length": 117.0, "completions/max_terminated_length": 146.0, "rewards/meter/mean": 0.9731119871139526, "rewards/meter/std": 0.013820521533489227, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8537229895591736, "rewards/repeat_soft/std": 0.0504949688911438, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.776397705078125, "rewards/total_composite/std": 0.026806162670254707, "reward": 0.776397705078125, "reward_std": 0.026806188747286797, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08743451535701752, "sampling/sampling_logp_difference/max": 2.394519090652466, "sampling/importance_sampling_ratio/min": 0.09121653437614441, "sampling/importance_sampling_ratio/mean": 1.0004037618637085, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4159902557730675, "clip_ratio/low_mean": 0.03675692854449153, "clip_ratio/low_min": 0.03675692854449153, "clip_ratio/high_mean": 0.04385602194815874, "clip_ratio/high_max": 0.04385602194815874, "clip_ratio/region_mean": 0.08061295049265027, "reward_total_mean": 0.776397705078125, "reward_meter_mean": 0.9731119871139526, "reward_meter_std": 0.013820521533489227, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8537229895591736, "reward_repeat_soft_std": 0.0504949688911438, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.776397705078125, "reward_total_composite_std": 0.026806162670254707} {"timestamp_utc": "2026-04-13T05:09:11Z", "mode": "train", "global_step": 2705, "epoch": 0.2717227523857358, "loss": 0.0053, "grad_norm": 3.661158800125122, "learning_rate": 1.8060606060606063e-06, "num_tokens": 5035512.0, "completions/mean_length": 141.125, "completions/min_length": 137.0, "completions/max_length": 148.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 141.125, "completions/min_terminated_length": 137.0, "completions/max_terminated_length": 148.0, "rewards/meter/mean": 0.9881019592285156, "rewards/meter/std": 0.004999305587261915, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6795893907546997, "rewards/repeat_soft/std": 0.08146630227565765, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7886048555374146, "rewards/total_composite/std": 0.007075757719576359, "reward": 0.7886048555374146, "reward_std": 0.007075760513544083, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06515192985534668, "sampling/sampling_logp_difference/max": 3.2770047187805176, "sampling/importance_sampling_ratio/min": 0.03774113208055496, "sampling/importance_sampling_ratio/mean": 0.9936439990997314, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2583769503980875, "clip_ratio/low_mean": 0.01584512274712324, "clip_ratio/low_min": 0.01584512274712324, "clip_ratio/high_mean": 0.03996527614071965, "clip_ratio/high_max": 0.03996527614071965, "clip_ratio/region_mean": 0.055810398887842894, "reward_total_mean": 0.7886048555374146, "reward_meter_mean": 0.9881019592285156, "reward_meter_std": 0.004999305587261915, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6795893907546997, "reward_repeat_soft_std": 0.08146630227565765, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7886048555374146, "reward_total_composite_std": 0.007075757719576359} {"timestamp_utc": "2026-04-13T05:09:18Z", "mode": "train", "global_step": 2706, "epoch": 0.2718232044198895, "loss": 0.0762, "grad_norm": 8.683341979980469, "learning_rate": 1.803030303030303e-06, "num_tokens": 5037730.0, "completions/mean_length": 106.25, "completions/min_length": 96.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.25, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.4325229525566101, "rewards/meter/std": 0.35765865445137024, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8452771902084351, "rewards/repeat_soft/std": 0.06503612548112869, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5551630854606628, "rewards/total_composite/std": 0.15868358314037323, "reward": 0.5551630854606628, "reward_std": 0.15868358314037323, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09588775038719177, "sampling/sampling_logp_difference/max": 3.5836496353149414, "sampling/importance_sampling_ratio/min": 0.0277741476893425, "sampling/importance_sampling_ratio/mean": 0.9985430836677551, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3440259024500847, "clip_ratio/low_mean": 0.04399640951305628, "clip_ratio/low_min": 0.04399640951305628, "clip_ratio/high_mean": 0.04338246490806341, "clip_ratio/high_max": 0.04338246490806341, "clip_ratio/region_mean": 0.08737887442111969, "reward_total_mean": 0.5551630854606628, "reward_meter_mean": 0.4325229525566101, "reward_meter_std": 0.35765865445137024, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8452771902084351, "reward_repeat_soft_std": 0.06503612548112869, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5551630854606628, "reward_total_composite_std": 0.15868358314037323} {"timestamp_utc": "2026-04-13T05:09:30Z", "mode": "train", "global_step": 2707, "epoch": 0.2719236564540432, "loss": -0.1399, "grad_norm": 2.146240711212158, "learning_rate": 1.8000000000000001e-06, "num_tokens": 5039415.0, "completions/mean_length": 114.625, "completions/min_length": 52.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 57.857147216796875, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9051285982131958, "rewards/meter/std": 0.164937362074852, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9791501760482788, "rewards/repeat_soft/std": 0.02762063965201378, "rewards/judge_quality/mean": 0.33249998092651367, "rewards/judge_quality/std": 0.1580009013414383, "rewards/total_composite/mean": 0.6695584654808044, "rewards/total_composite/std": 0.2849530875682831, "reward": 0.6695584654808044, "reward_std": 0.2849530875682831, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08926919102668762, "sampling/sampling_logp_difference/max": 2.363198757171631, "sampling/importance_sampling_ratio/min": 0.09411867707967758, "sampling/importance_sampling_ratio/mean": 0.9955125451087952, "sampling/importance_sampling_ratio/max": 1.8350598812103271, "entropy": 0.42866718396544456, "clip_ratio/low_mean": 0.008196720853447914, "clip_ratio/low_min": 0.008196720853447914, "clip_ratio/high_mean": 0.07644669245928526, "clip_ratio/high_max": 0.07644669245928526, "clip_ratio/region_mean": 0.08464341331273317, "reward_total_mean": 0.6695584654808044, "reward_meter_mean": 0.9051285982131958, "reward_meter_std": 0.164937362074852, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9791501760482788, "reward_repeat_soft_std": 0.02762063965201378, "reward_judge_quality_mean": 0.33249998092651367, "reward_judge_quality_std": 0.1580009013414383, "reward_total_composite_mean": 0.6695584654808044, "reward_total_composite_std": 0.2849530875682831} {"timestamp_utc": "2026-04-13T05:09:41Z", "mode": "train", "global_step": 2708, "epoch": 0.27202410848819686, "loss": -0.1904, "grad_norm": 1.9367650747299194, "learning_rate": 1.796969696969697e-06, "num_tokens": 5041457.0, "completions/mean_length": 148.25, "completions/min_length": 93.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 96.28572082519531, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.900143027305603, "rewards/meter/std": 0.21467316150665283, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9708433151245117, "rewards/repeat_soft/std": 0.030538037419319153, "rewards/judge_quality/mean": 0.4150000214576721, "rewards/judge_quality/std": 0.2436625361442566, "rewards/total_composite/mean": 0.7226165533065796, "rewards/total_composite/std": 0.29819852113723755, "reward": 0.7226165533065796, "reward_std": 0.29819852113723755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13274656236171722, "sampling/sampling_logp_difference/max": 6.080517768859863, "sampling/importance_sampling_ratio/min": 0.002286992035806179, "sampling/importance_sampling_ratio/mean": 0.9995225071907043, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5043321214616299, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.11087614251300693, "clip_ratio/high_max": 0.11087614251300693, "clip_ratio/region_mean": 0.11087614251300693, "reward_total_mean": 0.7226165533065796, "reward_meter_mean": 0.900143027305603, "reward_meter_std": 0.21467316150665283, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9708433151245117, "reward_repeat_soft_std": 0.030538037419319153, "reward_judge_quality_mean": 0.4150000214576721, "reward_judge_quality_std": 0.2436625361442566, "reward_total_composite_mean": 0.7226165533065796, "reward_total_composite_std": 0.29819852113723755} {"timestamp_utc": "2026-04-13T05:09:53Z", "mode": "train", "global_step": 2709, "epoch": 0.27212456052235057, "loss": -0.1058, "grad_norm": 2.5054869651794434, "learning_rate": 1.793939393939394e-06, "num_tokens": 5042850.0, "completions/mean_length": 91.125, "completions/min_length": 27.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 31.000001907348633, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9721280336380005, "rewards/meter/std": 0.05674280971288681, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.949787974357605, "rewards/repeat_soft/std": 0.03566979244351387, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.2876474857330322, "rewards/total_composite/mean": 0.7548753023147583, "rewards/total_composite/std": 0.3122991621494293, "reward": 0.7548753023147583, "reward_std": 0.3122991621494293, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1155344545841217, "sampling/sampling_logp_difference/max": 1.5676265954971313, "sampling/importance_sampling_ratio/min": 0.20853954553604126, "sampling/importance_sampling_ratio/mean": 1.007878065109253, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5510522201657295, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.11284509859979153, "clip_ratio/high_max": 0.11284509859979153, "clip_ratio/region_mean": 0.11284509859979153, "reward_total_mean": 0.7548753023147583, "reward_meter_mean": 0.9721280336380005, "reward_meter_std": 0.05674280971288681, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.949787974357605, "reward_repeat_soft_std": 0.03566979244351387, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.2876474857330322, "reward_total_composite_mean": 0.7548753023147583, "reward_total_composite_std": 0.3122991621494293} {"timestamp_utc": "2026-04-13T05:09:59Z", "mode": "train", "global_step": 2710, "epoch": 0.2722250125565043, "loss": 0.0528, "grad_norm": 8.090791702270508, "learning_rate": 1.7909090909090908e-06, "num_tokens": 5044404.0, "completions/mean_length": 51.25, "completions/min_length": 45.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.25, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9843956828117371, "rewards/meter/std": 0.00794876366853714, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.899063766002655, "rewards/repeat_soft/std": 0.07036890834569931, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.1011011004447937, "rewards/total_composite/mean": 0.8246344327926636, "rewards/total_composite/std": 0.03250264376401901, "reward": 0.8246344327926636, "reward_std": 0.03250264376401901, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07456564903259277, "sampling/sampling_logp_difference/max": 1.475923776626587, "sampling/importance_sampling_ratio/min": 0.22856749594211578, "sampling/importance_sampling_ratio/mean": 1.004401683807373, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39956579357385635, "clip_ratio/low_mean": 0.0572869717143476, "clip_ratio/low_min": 0.0572869717143476, "clip_ratio/high_mean": 0.021969697438180447, "clip_ratio/high_max": 0.021969697438180447, "clip_ratio/region_mean": 0.07925666915252805, "reward_total_mean": 0.8246344327926636, "reward_meter_mean": 0.9843956828117371, "reward_meter_std": 0.00794876366853714, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.899063766002655, "reward_repeat_soft_std": 0.07036890834569931, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.1011011004447937, "reward_total_composite_mean": 0.8246344327926636, "reward_total_composite_std": 0.03250264376401901} {"timestamp_utc": "2026-04-13T05:10:05Z", "mode": "train", "global_step": 2711, "epoch": 0.272325464590658, "loss": -0.0671, "grad_norm": 7.674440860748291, "learning_rate": 1.787878787878788e-06, "num_tokens": 5045800.0, "completions/mean_length": 51.5, "completions/min_length": 43.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.5, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9872823357582092, "rewards/meter/std": 0.004861865192651749, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8524351119995117, "rewards/repeat_soft/std": 0.07398942112922668, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.8178955912590027, "rewards/total_composite/std": 0.055165376514196396, "reward": 0.8178955912590027, "reward_std": 0.055165380239486694, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07976928353309631, "sampling/sampling_logp_difference/max": 1.1487007141113281, "sampling/importance_sampling_ratio/min": 0.31704843044281006, "sampling/importance_sampling_ratio/mean": 0.9930316209793091, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37581993266940117, "clip_ratio/low_mean": 0.048396961530670524, "clip_ratio/low_min": 0.048396961530670524, "clip_ratio/high_mean": 0.013888888992369175, "clip_ratio/high_max": 0.013888888992369175, "clip_ratio/region_mean": 0.0622858505230397, "reward_total_mean": 0.8178955912590027, "reward_meter_mean": 0.9872823357582092, "reward_meter_std": 0.004861865192651749, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8524351119995117, "reward_repeat_soft_std": 0.07398942112922668, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.8178955912590027, "reward_total_composite_std": 0.055165376514196396} {"timestamp_utc": "2026-04-13T05:10:12Z", "mode": "train", "global_step": 2712, "epoch": 0.27242591662481164, "loss": 0.0021, "grad_norm": 7.4585981369018555, "learning_rate": 1.7848484848484851e-06, "num_tokens": 5047516.0, "completions/mean_length": 59.5, "completions/min_length": 56.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.5, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9816904067993164, "rewards/meter/std": 0.006898047402501106, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9774754047393799, "rewards/repeat_soft/std": 0.02728254720568657, "rewards/judge_quality/mean": 0.7074999809265137, "rewards/judge_quality/std": 0.2474873960018158, "rewards/total_composite/mean": 0.9017582535743713, "rewards/total_composite/std": 0.0763942301273346, "reward": 0.9017582535743713, "reward_std": 0.0763942152261734, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.059436239302158356, "sampling/sampling_logp_difference/max": 1.0035347938537598, "sampling/importance_sampling_ratio/min": 0.36658138036727905, "sampling/importance_sampling_ratio/mean": 1.0085444450378418, "sampling/importance_sampling_ratio/max": 1.8474465608596802, "entropy": 0.2880927696824074, "clip_ratio/low_mean": 0.023660715203732252, "clip_ratio/low_min": 0.023660715203732252, "clip_ratio/high_mean": 0.051796937827020884, "clip_ratio/high_max": 0.051796937827020884, "clip_ratio/region_mean": 0.07545765303075314, "reward_total_mean": 0.9017582535743713, "reward_meter_mean": 0.9816904067993164, "reward_meter_std": 0.006898047402501106, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9774754047393799, "reward_repeat_soft_std": 0.02728254720568657, "reward_judge_quality_mean": 0.7074999809265137, "reward_judge_quality_std": 0.2474873960018158, "reward_total_composite_mean": 0.9017582535743713, "reward_total_composite_std": 0.0763942301273346} {"timestamp_utc": "2026-04-13T05:10:18Z", "mode": "train", "global_step": 2713, "epoch": 0.27252636865896535, "loss": 0.0639, "grad_norm": 13.804015159606934, "learning_rate": 1.781818181818182e-06, "num_tokens": 5049075.0, "completions/mean_length": 40.875, "completions/min_length": 34.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.875, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.857275664806366, "rewards/meter/std": 0.2990417182445526, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9406302571296692, "rewards/repeat_soft/std": 0.08223699033260345, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.7768371105194092, "rewards/total_composite/std": 0.14793996512889862, "reward": 0.7768371105194092, "reward_std": 0.14793995022773743, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09340512752532959, "sampling/sampling_logp_difference/max": 1.6099977493286133, "sampling/importance_sampling_ratio/min": 0.19988806545734406, "sampling/importance_sampling_ratio/mean": 1.0083321332931519, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5420490242540836, "clip_ratio/low_mean": 0.008152173832058907, "clip_ratio/low_min": 0.008152173832058907, "clip_ratio/high_mean": 0.10218337876722217, "clip_ratio/high_max": 0.10218337876722217, "clip_ratio/region_mean": 0.11033555259928107, "reward_total_mean": 0.7768371105194092, "reward_meter_mean": 0.857275664806366, "reward_meter_std": 0.2990417182445526, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9406302571296692, "reward_repeat_soft_std": 0.08223699033260345, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.7768371105194092, "reward_total_composite_std": 0.14793996512889862} {"timestamp_utc": "2026-04-13T05:10:29Z", "mode": "train", "global_step": 2714, "epoch": 0.27262682069311905, "loss": -0.2059, "grad_norm": 1.4448027610778809, "learning_rate": 1.778787878787879e-06, "num_tokens": 5051371.0, "completions/mean_length": 162.0, "completions/min_length": 105.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 112.00000762939453, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.9843789339065552, "rewards/meter/std": 0.020385170355439186, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7950543165206909, "rewards/repeat_soft/std": 0.12847861647605896, "rewards/judge_quality/mean": 0.30124998092651367, "rewards/judge_quality/std": 0.14913439750671387, "rewards/total_composite/mean": 0.6743602752685547, "rewards/total_composite/std": 0.27399376034736633, "reward": 0.6743602752685547, "reward_std": 0.27399376034736633, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11276665329933167, "sampling/sampling_logp_difference/max": 3.0110535621643066, "sampling/importance_sampling_ratio/min": 0.04923977702856064, "sampling/importance_sampling_ratio/mean": 0.996840238571167, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5174180418252945, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09835288673639297, "clip_ratio/high_max": 0.09835288673639297, "clip_ratio/region_mean": 0.09835288673639297, "reward_total_mean": 0.6743602752685547, "reward_meter_mean": 0.9843789339065552, "reward_meter_std": 0.020385170355439186, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7950543165206909, "reward_repeat_soft_std": 0.12847861647605896, "reward_judge_quality_mean": 0.30124998092651367, "reward_judge_quality_std": 0.14913439750671387, "reward_total_composite_mean": 0.6743602752685547, "reward_total_composite_std": 0.27399376034736633} {"timestamp_utc": "2026-04-13T05:10:36Z", "mode": "train", "global_step": 2715, "epoch": 0.2727272727272727, "loss": -0.0071, "grad_norm": 10.153380393981934, "learning_rate": 1.775757575757576e-06, "num_tokens": 5053106.0, "completions/mean_length": 56.875, "completions/min_length": 53.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.875, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9844064712524414, "rewards/meter/std": 0.01078957412391901, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9626266956329346, "rewards/repeat_soft/std": 0.02426435798406601, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8186205625534058, "rewards/total_composite/std": 0.0022853112313896418, "reward": 0.8186205625534058, "reward_std": 0.0022853133268654346, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09220621734857559, "sampling/sampling_logp_difference/max": 1.8302018642425537, "sampling/importance_sampling_ratio/min": 0.16038118302822113, "sampling/importance_sampling_ratio/mean": 1.000015139579773, "sampling/importance_sampling_ratio/max": 1.8058316707611084, "entropy": 0.46219685301184654, "clip_ratio/low_mean": 0.042632623575627804, "clip_ratio/low_min": 0.042632623575627804, "clip_ratio/high_mean": 0.0646758284419775, "clip_ratio/high_max": 0.0646758284419775, "clip_ratio/region_mean": 0.1073084520176053, "reward_total_mean": 0.8186205625534058, "reward_meter_mean": 0.9844064712524414, "reward_meter_std": 0.01078957412391901, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9626266956329346, "reward_repeat_soft_std": 0.02426435798406601, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8186205625534058, "reward_total_composite_std": 0.0022853112313896418} {"timestamp_utc": "2026-04-13T05:10:42Z", "mode": "train", "global_step": 2716, "epoch": 0.2728277247614264, "loss": 0.005, "grad_norm": 7.673124313354492, "learning_rate": 1.7727272727272729e-06, "num_tokens": 5054853.0, "completions/mean_length": 70.375, "completions/min_length": 67.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.375, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.9863662719726562, "rewards/meter/std": 0.006157987751066685, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9293677806854248, "rewards/repeat_soft/std": 0.04843641817569733, "rewards/judge_quality/mean": 0.6812499761581421, "rewards/judge_quality/std": 0.25542333722114563, "rewards/total_composite/mean": 0.8911765813827515, "rewards/total_composite/std": 0.07472305744886398, "reward": 0.8911765813827515, "reward_std": 0.07472305744886398, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06744091212749481, "sampling/sampling_logp_difference/max": 2.7103333473205566, "sampling/importance_sampling_ratio/min": 0.06651463359594345, "sampling/importance_sampling_ratio/mean": 1.0083823204040527, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31821873411536217, "clip_ratio/low_mean": 0.023422847967594862, "clip_ratio/low_min": 0.023422847967594862, "clip_ratio/high_mean": 0.026114049833267927, "clip_ratio/high_max": 0.026114049833267927, "clip_ratio/region_mean": 0.04953689780086279, "reward_total_mean": 0.8911765813827515, "reward_meter_mean": 0.9863662719726562, "reward_meter_std": 0.006157987751066685, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9293677806854248, "reward_repeat_soft_std": 0.04843641817569733, "reward_judge_quality_mean": 0.6812499761581421, "reward_judge_quality_std": 0.25542333722114563, "reward_total_composite_mean": 0.8911765813827515, "reward_total_composite_std": 0.07472305744886398} {"timestamp_utc": "2026-04-13T05:10:49Z", "mode": "train", "global_step": 2717, "epoch": 0.2729281767955801, "loss": 0.0547, "grad_norm": 9.631240844726562, "learning_rate": 1.76969696969697e-06, "num_tokens": 5056627.0, "completions/mean_length": 63.75, "completions/min_length": 57.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.75, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9661099910736084, "rewards/meter/std": 0.043430935591459274, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9744170308113098, "rewards/repeat_soft/std": 0.025087647140026093, "rewards/judge_quality/mean": 0.45249998569488525, "rewards/judge_quality/std": 0.06902381032705307, "rewards/total_composite/mean": 0.8179411888122559, "rewards/total_composite/std": 0.0318836085498333, "reward": 0.8179411888122559, "reward_std": 0.031883593648672104, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10081133991479874, "sampling/sampling_logp_difference/max": 1.1250276565551758, "sampling/importance_sampling_ratio/min": 0.32464346289634705, "sampling/importance_sampling_ratio/mean": 1.0249276161193848, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6157990843057632, "clip_ratio/low_mean": 0.053593436256051064, "clip_ratio/low_min": 0.053593436256051064, "clip_ratio/high_mean": 0.04583297669887543, "clip_ratio/high_max": 0.04583297669887543, "clip_ratio/region_mean": 0.09942641295492649, "reward_total_mean": 0.8179411888122559, "reward_meter_mean": 0.9661099910736084, "reward_meter_std": 0.043430935591459274, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9744170308113098, "reward_repeat_soft_std": 0.025087647140026093, "reward_judge_quality_mean": 0.45249998569488525, "reward_judge_quality_std": 0.06902381032705307, "reward_total_composite_mean": 0.8179411888122559, "reward_total_composite_std": 0.0318836085498333} {"timestamp_utc": "2026-04-13T05:10:56Z", "mode": "train", "global_step": 2718, "epoch": 0.2730286288297338, "loss": -0.0393, "grad_norm": 5.686733245849609, "learning_rate": 1.7666666666666668e-06, "num_tokens": 5058947.0, "completions/mean_length": 120.0, "completions/min_length": 101.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.0, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.9946779012680054, "rewards/meter/std": 0.0012321420945227146, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8829200267791748, "rewards/repeat_soft/std": 0.036797598004341125, "rewards/judge_quality/mean": 0.5575000047683716, "rewards/judge_quality/std": 0.19955310225486755, "rewards/total_composite/mean": 0.8343970775604248, "rewards/total_composite/std": 0.07668539136648178, "reward": 0.8343970775604248, "reward_std": 0.07668539881706238, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07149680703878403, "sampling/sampling_logp_difference/max": 1.3798167705535889, "sampling/importance_sampling_ratio/min": 0.2516246736049652, "sampling/importance_sampling_ratio/mean": 1.007676601409912, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40924064442515373, "clip_ratio/low_mean": 0.03829572023823857, "clip_ratio/low_min": 0.03829572023823857, "clip_ratio/high_mean": 0.025421466678380966, "clip_ratio/high_max": 0.025421466678380966, "clip_ratio/region_mean": 0.06371718691661954, "reward_total_mean": 0.8343970775604248, "reward_meter_mean": 0.9946779012680054, "reward_meter_std": 0.0012321420945227146, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8829200267791748, "reward_repeat_soft_std": 0.036797598004341125, "reward_judge_quality_mean": 0.5575000047683716, "reward_judge_quality_std": 0.19955310225486755, "reward_total_composite_mean": 0.8343970775604248, "reward_total_composite_std": 0.07668539136648178} {"timestamp_utc": "2026-04-13T05:11:02Z", "mode": "train", "global_step": 2719, "epoch": 0.2731290808638875, "loss": 0.0293, "grad_norm": 9.07138729095459, "learning_rate": 1.7636363636363638e-06, "num_tokens": 5060592.0, "completions/mean_length": 53.625, "completions/min_length": 50.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.625, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9878568649291992, "rewards/meter/std": 0.004308065865188837, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.990628719329834, "rewards/repeat_soft/std": 0.01406632736325264, "rewards/judge_quality/mean": 0.6487500071525574, "rewards/judge_quality/std": 0.2507951855659485, "rewards/total_composite/mean": 0.8882234692573547, "rewards/total_composite/std": 0.07542760670185089, "reward": 0.8882234692573547, "reward_std": 0.07542760670185089, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1033850908279419, "sampling/sampling_logp_difference/max": 1.4964828491210938, "sampling/importance_sampling_ratio/min": 0.23727086186408997, "sampling/importance_sampling_ratio/mean": 1.0058350563049316, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4701782837510109, "clip_ratio/low_mean": 0.04642732348293066, "clip_ratio/low_min": 0.04642732348293066, "clip_ratio/high_mean": 0.05153378564864397, "clip_ratio/high_max": 0.05153378564864397, "clip_ratio/region_mean": 0.09796110913157463, "reward_total_mean": 0.8882234692573547, "reward_meter_mean": 0.9878568649291992, "reward_meter_std": 0.004308065865188837, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.990628719329834, "reward_repeat_soft_std": 0.01406632736325264, "reward_judge_quality_mean": 0.6487500071525574, "reward_judge_quality_std": 0.2507951855659485, "reward_total_composite_mean": 0.8882234692573547, "reward_total_composite_std": 0.07542760670185089} {"timestamp_utc": "2026-04-13T05:11:11Z", "mode": "train", "global_step": 2720, "epoch": 0.2732295328980412, "loss": 0.0211, "grad_norm": 4.012598037719727, "learning_rate": 1.7606060606060606e-06, "num_tokens": 5063840.0, "completions/mean_length": 195.0, "completions/min_length": 164.0, "completions/max_length": 234.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 195.0, "completions/min_terminated_length": 164.0, "completions/max_terminated_length": 234.0, "rewards/meter/mean": 0.9910176992416382, "rewards/meter/std": 0.0067486269399523735, "rewards/count_adherence/mean": 0.8958333134651184, "rewards/count_adherence/std": 0.08625821024179459, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.760661244392395, "rewards/repeat_soft/std": 0.1430862993001938, "rewards/judge_quality/mean": 0.3137499988079071, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.75052410364151, "rewards/total_composite/std": 0.02884608693420887, "reward": 0.75052410364151, "reward_std": 0.02884609065949917, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0681898221373558, "sampling/sampling_logp_difference/max": 2.288301944732666, "sampling/importance_sampling_ratio/min": 0.10143856704235077, "sampling/importance_sampling_ratio/mean": 1.0063782930374146, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.324303824454546, "clip_ratio/low_mean": 0.02450370555743575, "clip_ratio/low_min": 0.02450370555743575, "clip_ratio/high_mean": 0.0334926787763834, "clip_ratio/high_max": 0.0334926787763834, "clip_ratio/region_mean": 0.05799638433381915, "reward_total_mean": 0.75052410364151, "reward_meter_mean": 0.9910176992416382, "reward_meter_std": 0.0067486269399523735, "reward_count_adherence_mean": 0.8958333134651184, "reward_count_adherence_std": 0.08625821024179459, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.760661244392395, "reward_repeat_soft_std": 0.1430862993001938, "reward_judge_quality_mean": 0.3137499988079071, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.75052410364151, "reward_total_composite_std": 0.02884608693420887} {"timestamp_utc": "2026-04-13T05:11:22Z", "mode": "train", "global_step": 2721, "epoch": 0.2733299849321949, "loss": -0.1261, "grad_norm": 2.1725642681121826, "learning_rate": 1.7575757575757577e-06, "num_tokens": 5065508.0, "completions/mean_length": 172.5, "completions/min_length": 54.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 59.333335876464844, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.6953457593917847, "rewards/meter/std": 0.39587533473968506, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9855368137359619, "rewards/repeat_soft/std": 0.013913910835981369, "rewards/judge_quality/mean": 0.2787500023841858, "rewards/judge_quality/std": 0.17381742596626282, "rewards/total_composite/mean": 0.5712400674819946, "rewards/total_composite/std": 0.36099672317504883, "reward": 0.5712400674819946, "reward_std": 0.36099672317504883, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13010713458061218, "sampling/sampling_logp_difference/max": 1.7770769596099854, "sampling/importance_sampling_ratio/min": 0.169131800532341, "sampling/importance_sampling_ratio/mean": 0.9985567927360535, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47488562762737274, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08886591251939535, "clip_ratio/high_max": 0.08886591251939535, "clip_ratio/region_mean": 0.08886591251939535, "reward_total_mean": 0.5712400674819946, "reward_meter_mean": 0.6953457593917847, "reward_meter_std": 0.39587533473968506, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9855368137359619, "reward_repeat_soft_std": 0.013913910835981369, "reward_judge_quality_mean": 0.2787500023841858, "reward_judge_quality_std": 0.17381742596626282, "reward_total_composite_mean": 0.5712400674819946, "reward_total_composite_std": 0.36099672317504883} {"timestamp_utc": "2026-04-13T05:11:29Z", "mode": "train", "global_step": 2722, "epoch": 0.27343043696634856, "loss": 0.0336, "grad_norm": 10.934043884277344, "learning_rate": 1.7545454545454545e-06, "num_tokens": 5067730.0, "completions/mean_length": 115.75, "completions/min_length": 109.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.75, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.9905948042869568, "rewards/meter/std": 0.0037175663746893406, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9032509326934814, "rewards/repeat_soft/std": 0.07930579036474228, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8120927810668945, "rewards/total_composite/std": 0.008077739737927914, "reward": 0.8120927810668945, "reward_std": 0.008077748119831085, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07999104261398315, "sampling/sampling_logp_difference/max": 1.8589916229248047, "sampling/importance_sampling_ratio/min": 0.15582968294620514, "sampling/importance_sampling_ratio/mean": 0.9925107359886169, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3469668962061405, "clip_ratio/low_mean": 0.022308197803795338, "clip_ratio/low_min": 0.022308197803795338, "clip_ratio/high_mean": 0.05746741546317935, "clip_ratio/high_max": 0.05746741546317935, "clip_ratio/region_mean": 0.07977561326697469, "reward_total_mean": 0.8120927810668945, "reward_meter_mean": 0.9905948042869568, "reward_meter_std": 0.0037175663746893406, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9032509326934814, "reward_repeat_soft_std": 0.07930579036474228, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8120927810668945, "reward_total_composite_std": 0.008077739737927914} {"timestamp_utc": "2026-04-13T05:11:35Z", "mode": "train", "global_step": 2723, "epoch": 0.27353088900050226, "loss": 0.1068, "grad_norm": 18.181842803955078, "learning_rate": 1.7515151515151516e-06, "num_tokens": 5069554.0, "completions/mean_length": 67.0, "completions/min_length": 52.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.762507438659668, "rewards/meter/std": 0.25989288091659546, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9896728992462158, "rewards/repeat_soft/std": 0.014009211212396622, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.7402206659317017, "rewards/total_composite/std": 0.14042887091636658, "reward": 0.7402206659317017, "reward_std": 0.14042888581752777, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11705783754587173, "sampling/sampling_logp_difference/max": 3.054142475128174, "sampling/importance_sampling_ratio/min": 0.04716314747929573, "sampling/importance_sampling_ratio/mean": 1.0182735919952393, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4871355630457401, "clip_ratio/low_mean": 0.031261726282536983, "clip_ratio/low_min": 0.031261726282536983, "clip_ratio/high_mean": 0.0556424087844789, "clip_ratio/high_max": 0.0556424087844789, "clip_ratio/region_mean": 0.08690413506701589, "reward_total_mean": 0.7402206659317017, "reward_meter_mean": 0.762507438659668, "reward_meter_std": 0.25989288091659546, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9896728992462158, "reward_repeat_soft_std": 0.014009211212396622, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.7402206659317017, "reward_total_composite_std": 0.14042887091636658} {"timestamp_utc": "2026-04-13T05:11:47Z", "mode": "train", "global_step": 2724, "epoch": 0.273631341034656, "loss": -0.2273, "grad_norm": 1.5717029571533203, "learning_rate": 1.7484848484848486e-06, "num_tokens": 5071923.0, "completions/mean_length": 192.125, "completions/min_length": 139.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 146.42857360839844, "completions/min_terminated_length": 139.0, "completions/max_terminated_length": 156.0, "rewards/meter/mean": 0.9252983331680298, "rewards/meter/std": 0.11701630800962448, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9659644365310669, "rewards/repeat_soft/std": 0.012699170969426632, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.13715866208076477, "rewards/total_composite/mean": 0.6829851269721985, "rewards/total_composite/std": 0.28030896186828613, "reward": 0.6829851269721985, "reward_std": 0.28030896186828613, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08195871114730835, "sampling/sampling_logp_difference/max": 2.53867244720459, "sampling/importance_sampling_ratio/min": 0.07897116988897324, "sampling/importance_sampling_ratio/mean": 1.0050290822982788, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31926513463258743, "clip_ratio/low_mean": 0.008992806077003479, "clip_ratio/low_min": 0.008992806077003479, "clip_ratio/high_mean": 0.04775674291886389, "clip_ratio/high_max": 0.04775674291886389, "clip_ratio/region_mean": 0.05674954899586737, "reward_total_mean": 0.6829851269721985, "reward_meter_mean": 0.9252983331680298, "reward_meter_std": 0.11701630800962448, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9659644365310669, "reward_repeat_soft_std": 0.012699170969426632, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.13715866208076477, "reward_total_composite_mean": 0.6829851269721985, "reward_total_composite_std": 0.28030896186828613} {"timestamp_utc": "2026-04-13T05:11:55Z", "mode": "train", "global_step": 2725, "epoch": 0.2737317930688096, "loss": -0.0191, "grad_norm": 10.08575439453125, "learning_rate": 1.7454545454545456e-06, "num_tokens": 5074660.0, "completions/mean_length": 144.125, "completions/min_length": 133.0, "completions/max_length": 164.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 144.125, "completions/min_terminated_length": 133.0, "completions/max_terminated_length": 164.0, "rewards/meter/mean": 0.9559518098831177, "rewards/meter/std": 0.10157521069049835, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7974700331687927, "rewards/repeat_soft/std": 0.10458426922559738, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7668002843856812, "rewards/total_composite/std": 0.04746891185641289, "reward": 0.7668002843856812, "reward_std": 0.047468896955251694, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07638435810804367, "sampling/sampling_logp_difference/max": 1.7692124843597412, "sampling/importance_sampling_ratio/min": 0.17046718299388885, "sampling/importance_sampling_ratio/mean": 1.0066947937011719, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3795163445174694, "clip_ratio/low_mean": 0.03338650846853852, "clip_ratio/low_min": 0.03338650846853852, "clip_ratio/high_mean": 0.04662961233407259, "clip_ratio/high_max": 0.04662961233407259, "clip_ratio/region_mean": 0.08001612080261111, "reward_total_mean": 0.7668002843856812, "reward_meter_mean": 0.9559518098831177, "reward_meter_std": 0.10157521069049835, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7974700331687927, "reward_repeat_soft_std": 0.10458426922559738, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7668002843856812, "reward_total_composite_std": 0.04746891185641289} {"timestamp_utc": "2026-04-13T05:12:01Z", "mode": "train", "global_step": 2726, "epoch": 0.27383224510296333, "loss": 0.0159, "grad_norm": 7.943268775939941, "learning_rate": 1.7424242424242427e-06, "num_tokens": 5076376.0, "completions/mean_length": 58.5, "completions/min_length": 54.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.5, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.8479034304618835, "rewards/meter/std": 0.252631813287735, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8992248773574829, "rewards/repeat_soft/std": 0.05451371148228645, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.7684789896011353, "rewards/total_composite/std": 0.06776487827301025, "reward": 0.7684789896011353, "reward_std": 0.06776487827301025, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08133484423160553, "sampling/sampling_logp_difference/max": 2.2027034759521484, "sampling/importance_sampling_ratio/min": 0.11050400882959366, "sampling/importance_sampling_ratio/mean": 0.9976280331611633, "sampling/importance_sampling_ratio/max": 1.8398507833480835, "entropy": 0.3501460663974285, "clip_ratio/low_mean": 0.014893149491399527, "clip_ratio/low_min": 0.014893149491399527, "clip_ratio/high_mean": 0.037807958433404565, "clip_ratio/high_max": 0.037807958433404565, "clip_ratio/region_mean": 0.05270110792480409, "reward_total_mean": 0.7684789896011353, "reward_meter_mean": 0.8479034304618835, "reward_meter_std": 0.252631813287735, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8992248773574829, "reward_repeat_soft_std": 0.05451371148228645, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.7684789896011353, "reward_total_composite_std": 0.06776487827301025} {"timestamp_utc": "2026-04-13T05:12:13Z", "mode": "train", "global_step": 2727, "epoch": 0.27393269713711704, "loss": -0.1722, "grad_norm": 2.6763200759887695, "learning_rate": 1.7393939393939397e-06, "num_tokens": 5078617.0, "completions/mean_length": 178.125, "completions/min_length": 117.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 130.42857360839844, "completions/min_terminated_length": 117.0, "completions/max_terminated_length": 153.0, "rewards/meter/mean": 0.4985562264919281, "rewards/meter/std": 0.363071471452713, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8731704950332642, "rewards/repeat_soft/std": 0.07792337983846664, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.5134633779525757, "rewards/total_composite/std": 0.25916799902915955, "reward": 0.5134633779525757, "reward_std": 0.25916796922683716, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09499241411685944, "sampling/sampling_logp_difference/max": 1.718159794807434, "sampling/importance_sampling_ratio/min": 0.17939597368240356, "sampling/importance_sampling_ratio/mean": 1.0094648599624634, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3369644843041897, "clip_ratio/low_mean": 0.043591524474322796, "clip_ratio/low_min": 0.043591524474322796, "clip_ratio/high_mean": 0.04163744766265154, "clip_ratio/high_max": 0.04163744766265154, "clip_ratio/region_mean": 0.08522897213697433, "reward_total_mean": 0.5134633779525757, "reward_meter_mean": 0.4985562264919281, "reward_meter_std": 0.363071471452713, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8731704950332642, "reward_repeat_soft_std": 0.07792337983846664, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.5134633779525757, "reward_total_composite_std": 0.25916799902915955} {"timestamp_utc": "2026-04-13T05:12:21Z", "mode": "train", "global_step": 2728, "epoch": 0.2740331491712707, "loss": 0.0114, "grad_norm": 4.823965549468994, "learning_rate": 1.7363636363636366e-06, "num_tokens": 5082016.0, "completions/mean_length": 200.875, "completions/min_length": 182.0, "completions/max_length": 211.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 200.875, "completions/min_terminated_length": 182.0, "completions/max_terminated_length": 211.0, "rewards/meter/mean": 0.9950728416442871, "rewards/meter/std": 0.0017217440763488412, "rewards/count_adherence/mean": 0.8958333134651184, "rewards/count_adherence/std": 0.08625820279121399, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8957263231277466, "rewards/repeat_soft/std": 0.03052685409784317, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.8164803981781006, "rewards/total_composite/std": 0.050875429064035416, "reward": 0.8164803981781006, "reward_std": 0.050875432789325714, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07398919761180878, "sampling/sampling_logp_difference/max": 1.6341863870620728, "sampling/importance_sampling_ratio/min": 0.19511105120182037, "sampling/importance_sampling_ratio/mean": 1.003313660621643, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37800271809101105, "clip_ratio/low_mean": 0.052232079673558474, "clip_ratio/low_min": 0.052232079673558474, "clip_ratio/high_mean": 0.008207070641219616, "clip_ratio/high_max": 0.008207070641219616, "clip_ratio/region_mean": 0.06043915031477809, "reward_total_mean": 0.8164803981781006, "reward_meter_mean": 0.9950728416442871, "reward_meter_std": 0.0017217440763488412, "reward_count_adherence_mean": 0.8958333134651184, "reward_count_adherence_std": 0.08625820279121399, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8957263231277466, "reward_repeat_soft_std": 0.03052685409784317, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.8164803981781006, "reward_total_composite_std": 0.050875429064035416} {"timestamp_utc": "2026-04-13T05:12:29Z", "mode": "train", "global_step": 2729, "epoch": 0.2741336012054244, "loss": 0.0861, "grad_norm": 6.352367877960205, "learning_rate": 1.7333333333333336e-06, "num_tokens": 5084796.0, "completions/mean_length": 151.5, "completions/min_length": 131.0, "completions/max_length": 180.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 151.5, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 180.0, "rewards/meter/mean": 0.8841935396194458, "rewards/meter/std": 0.2565622329711914, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8561599850654602, "rewards/repeat_soft/std": 0.052113331854343414, "rewards/judge_quality/mean": 0.3137499690055847, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7213781476020813, "rewards/total_composite/std": 0.1363220363855362, "reward": 0.7213781476020813, "reward_std": 0.1363220512866974, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09478788822889328, "sampling/sampling_logp_difference/max": 8.769278526306152, "sampling/importance_sampling_ratio/min": 0.00015543567133136094, "sampling/importance_sampling_ratio/mean": 1.0020617246627808, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3987192288041115, "clip_ratio/low_mean": 0.008333333767950535, "clip_ratio/low_min": 0.008333333767950535, "clip_ratio/high_mean": 0.0772048020735383, "clip_ratio/high_max": 0.0772048020735383, "clip_ratio/region_mean": 0.08553813584148884, "reward_total_mean": 0.7213781476020813, "reward_meter_mean": 0.8841935396194458, "reward_meter_std": 0.2565622329711914, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8561599850654602, "reward_repeat_soft_std": 0.052113331854343414, "reward_judge_quality_mean": 0.3137499690055847, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7213781476020813, "reward_total_composite_std": 0.1363220363855362} {"timestamp_utc": "2026-04-13T05:12:36Z", "mode": "train", "global_step": 2730, "epoch": 0.2742340532395781, "loss": 0.0054, "grad_norm": 7.319043159484863, "learning_rate": 1.7303030303030304e-06, "num_tokens": 5086576.0, "completions/mean_length": 67.5, "completions/min_length": 64.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.5, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9933198690414429, "rewards/meter/std": 0.0027113333344459534, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9491604566574097, "rewards/repeat_soft/std": 0.047776468098163605, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.10260014235973358, "rewards/total_composite/mean": 0.8325349688529968, "rewards/total_composite/std": 0.02988884598016739, "reward": 0.8325349688529968, "reward_std": 0.029888847842812538, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09549494087696075, "sampling/sampling_logp_difference/max": 1.7607998847961426, "sampling/importance_sampling_ratio/min": 0.17190730571746826, "sampling/importance_sampling_ratio/mean": 0.9977509379386902, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4420680068433285, "clip_ratio/low_mean": 0.05799241946078837, "clip_ratio/low_min": 0.05799241946078837, "clip_ratio/high_mean": 0.014492753893136978, "clip_ratio/high_max": 0.014492753893136978, "clip_ratio/region_mean": 0.07248517335392535, "reward_total_mean": 0.8325349688529968, "reward_meter_mean": 0.9933198690414429, "reward_meter_std": 0.0027113333344459534, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9491604566574097, "reward_repeat_soft_std": 0.047776468098163605, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.10260014235973358, "reward_total_composite_mean": 0.8325349688529968, "reward_total_composite_std": 0.02988884598016739} {"timestamp_utc": "2026-04-13T05:12:42Z", "mode": "train", "global_step": 2731, "epoch": 0.27433450527373177, "loss": 0.0538, "grad_norm": 11.664392471313477, "learning_rate": 1.7272727272727275e-06, "num_tokens": 5088249.0, "completions/mean_length": 49.125, "completions/min_length": 42.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.125, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.9789092540740967, "rewards/meter/std": 0.009124401025474072, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.930320143699646, "rewards/repeat_soft/std": 0.06018800660967827, "rewards/judge_quality/mean": 0.6600000262260437, "rewards/judge_quality/std": 0.23384979367256165, "rewards/total_composite/mean": 0.8815411925315857, "rewards/total_composite/std": 0.06681161373853683, "reward": 0.8815411925315857, "reward_std": 0.06681162863969803, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10608694702386856, "sampling/sampling_logp_difference/max": 1.6675701141357422, "sampling/importance_sampling_ratio/min": 0.18870504200458527, "sampling/importance_sampling_ratio/mean": 1.0024670362472534, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.431756142526865, "clip_ratio/low_mean": 0.049878387711942196, "clip_ratio/low_min": 0.049878387711942196, "clip_ratio/high_mean": 0.04856570530682802, "clip_ratio/high_max": 0.04856570530682802, "clip_ratio/region_mean": 0.09844409301877022, "reward_total_mean": 0.8815411925315857, "reward_meter_mean": 0.9789092540740967, "reward_meter_std": 0.009124401025474072, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.930320143699646, "reward_repeat_soft_std": 0.06018800660967827, "reward_judge_quality_mean": 0.6600000262260437, "reward_judge_quality_std": 0.23384979367256165, "reward_total_composite_mean": 0.8815411925315857, "reward_total_composite_std": 0.06681161373853683} {"timestamp_utc": "2026-04-13T05:12:53Z", "mode": "train", "global_step": 2732, "epoch": 0.2744349573078855, "loss": -0.1834, "grad_norm": 1.7759878635406494, "learning_rate": 1.7242424242424243e-06, "num_tokens": 5090079.0, "completions/mean_length": 140.75, "completions/min_length": 71.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 87.71428680419922, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.9630715847015381, "rewards/meter/std": 0.03198624774813652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9591677784919739, "rewards/repeat_soft/std": 0.03111003153026104, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.23445606231689453, "rewards/total_composite/mean": 0.7252597808837891, "rewards/total_composite/std": 0.29849496483802795, "reward": 0.7252597808837891, "reward_std": 0.29849493503570557, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09574726969003677, "sampling/sampling_logp_difference/max": 1.947042465209961, "sampling/importance_sampling_ratio/min": 0.14269547164440155, "sampling/importance_sampling_ratio/mean": 0.9970415234565735, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.35438019409775734, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0945387976244092, "clip_ratio/high_max": 0.0945387976244092, "clip_ratio/region_mean": 0.0945387976244092, "reward_total_mean": 0.7252597808837891, "reward_meter_mean": 0.9630715847015381, "reward_meter_std": 0.03198624774813652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9591677784919739, "reward_repeat_soft_std": 0.03111003153026104, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.23445606231689453, "reward_total_composite_mean": 0.7252597808837891, "reward_total_composite_std": 0.29849496483802795} {"timestamp_utc": "2026-04-13T05:13:00Z", "mode": "train", "global_step": 2733, "epoch": 0.2745354093420392, "loss": 0.0054, "grad_norm": 8.54227066040039, "learning_rate": 1.7212121212121214e-06, "num_tokens": 5092068.0, "completions/mean_length": 63.625, "completions/min_length": 58.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.625, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.8520309925079346, "rewards/meter/std": 0.24960999190807343, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.970098614692688, "rewards/repeat_soft/std": 0.017413537949323654, "rewards/judge_quality/mean": 0.5612499713897705, "rewards/judge_quality/std": 0.1968638300895691, "rewards/total_composite/mean": 0.7987987995147705, "rewards/total_composite/std": 0.12187225371599197, "reward": 0.7987987995147705, "reward_std": 0.12187224626541138, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09758984297513962, "sampling/sampling_logp_difference/max": 1.5222355127334595, "sampling/importance_sampling_ratio/min": 0.21822351217269897, "sampling/importance_sampling_ratio/mean": 1.0085747241973877, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5082843825221062, "clip_ratio/low_mean": 0.0310469763353467, "clip_ratio/low_min": 0.0310469763353467, "clip_ratio/high_mean": 0.06576588144525886, "clip_ratio/high_max": 0.06576588144525886, "clip_ratio/region_mean": 0.09681285778060555, "reward_total_mean": 0.7987987995147705, "reward_meter_mean": 0.8520309925079346, "reward_meter_std": 0.24960999190807343, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.970098614692688, "reward_repeat_soft_std": 0.017413537949323654, "reward_judge_quality_mean": 0.5612499713897705, "reward_judge_quality_std": 0.1968638300895691, "reward_total_composite_mean": 0.7987987995147705, "reward_total_composite_std": 0.12187225371599197} {"timestamp_utc": "2026-04-13T05:13:07Z", "mode": "train", "global_step": 2734, "epoch": 0.2746358613761929, "loss": -0.0765, "grad_norm": 7.158923625946045, "learning_rate": 1.7181818181818182e-06, "num_tokens": 5094579.0, "completions/mean_length": 132.875, "completions/min_length": 112.0, "completions/max_length": 156.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.875, "completions/min_terminated_length": 112.0, "completions/max_terminated_length": 156.0, "rewards/meter/mean": 0.9935674667358398, "rewards/meter/std": 0.0029393730219453573, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.915935754776001, "rewards/repeat_soft/std": 0.03876753896474838, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7912614345550537, "rewards/total_composite/std": 0.01970824785530567, "reward": 0.7912614345550537, "reward_std": 0.019708259031176567, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08684112131595612, "sampling/sampling_logp_difference/max": 1.532597541809082, "sampling/importance_sampling_ratio/min": 0.21597394347190857, "sampling/importance_sampling_ratio/mean": 1.0074690580368042, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.46499888598918915, "clip_ratio/low_mean": 0.05585706466808915, "clip_ratio/low_min": 0.05585706466808915, "clip_ratio/high_mean": 0.03243569936603308, "clip_ratio/high_max": 0.03243569936603308, "clip_ratio/region_mean": 0.08829276403412223, "reward_total_mean": 0.7912614345550537, "reward_meter_mean": 0.9935674667358398, "reward_meter_std": 0.0029393730219453573, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.915935754776001, "reward_repeat_soft_std": 0.03876753896474838, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7912614345550537, "reward_total_composite_std": 0.01970824785530567} {"timestamp_utc": "2026-04-13T05:13:14Z", "mode": "train", "global_step": 2735, "epoch": 0.27473631341034654, "loss": 0.0431, "grad_norm": 9.302800178527832, "learning_rate": 1.7151515151515152e-06, "num_tokens": 5096512.0, "completions/mean_length": 88.625, "completions/min_length": 82.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 88.625, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.7387542724609375, "rewards/meter/std": 0.3291255235671997, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8812230229377747, "rewards/repeat_soft/std": 0.07583532482385635, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.7340617179870605, "rewards/total_composite/std": 0.16788791120052338, "reward": 0.7340617179870605, "reward_std": 0.16788791120052338, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09560149163007736, "sampling/sampling_logp_difference/max": 1.2632360458374023, "sampling/importance_sampling_ratio/min": 0.2827375829219818, "sampling/importance_sampling_ratio/mean": 1.0141483545303345, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4474288523197174, "clip_ratio/low_mean": 0.022109460085630417, "clip_ratio/low_min": 0.022109460085630417, "clip_ratio/high_mean": 0.0759030245244503, "clip_ratio/high_max": 0.0759030245244503, "clip_ratio/region_mean": 0.09801248461008072, "reward_total_mean": 0.7340617179870605, "reward_meter_mean": 0.7387542724609375, "reward_meter_std": 0.3291255235671997, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8812230229377747, "reward_repeat_soft_std": 0.07583532482385635, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.7340617179870605, "reward_total_composite_std": 0.16788791120052338} {"timestamp_utc": "2026-04-13T05:13:20Z", "mode": "train", "global_step": 2736, "epoch": 0.27483676544450025, "loss": 0.0035, "grad_norm": 7.698275566101074, "learning_rate": 1.7121212121212123e-06, "num_tokens": 5098224.0, "completions/mean_length": 61.0, "completions/min_length": 58.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9220463037490845, "rewards/meter/std": 0.1410762518644333, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9509108066558838, "rewards/repeat_soft/std": 0.028308043256402016, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.8797619342803955, "rewards/total_composite/std": 0.11900720745325089, "reward": 0.8797619342803955, "reward_std": 0.11900721490383148, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09131740033626556, "sampling/sampling_logp_difference/max": 1.323699951171875, "sampling/importance_sampling_ratio/min": 0.26614874601364136, "sampling/importance_sampling_ratio/mean": 1.0026592016220093, "sampling/importance_sampling_ratio/max": 1.8553111553192139, "entropy": 0.5535133369266987, "clip_ratio/low_mean": 0.034742419607937336, "clip_ratio/low_min": 0.034742419607937336, "clip_ratio/high_mean": 0.04771327134221792, "clip_ratio/high_max": 0.04771327134221792, "clip_ratio/region_mean": 0.08245569095015526, "reward_total_mean": 0.8797619342803955, "reward_meter_mean": 0.9220463037490845, "reward_meter_std": 0.1410762518644333, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9509108066558838, "reward_repeat_soft_std": 0.028308043256402016, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.8797619342803955, "reward_total_composite_std": 0.11900720745325089} {"timestamp_utc": "2026-04-13T05:13:27Z", "mode": "train", "global_step": 2737, "epoch": 0.27493721747865396, "loss": 0.0361, "grad_norm": 7.5961594581604, "learning_rate": 1.7090909090909091e-06, "num_tokens": 5100081.0, "completions/mean_length": 74.125, "completions/min_length": 70.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.125, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.8960349559783936, "rewards/meter/std": 0.186055526137352, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9511100649833679, "rewards/repeat_soft/std": 0.03043997846543789, "rewards/judge_quality/mean": 0.6274999976158142, "rewards/judge_quality/std": 0.2290196716785431, "rewards/total_composite/mean": 0.8365767002105713, "rewards/total_composite/std": 0.13514691591262817, "reward": 0.8365767002105713, "reward_std": 0.13514691591262817, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08638309687376022, "sampling/sampling_logp_difference/max": 2.0133485794067383, "sampling/importance_sampling_ratio/min": 0.13354074954986572, "sampling/importance_sampling_ratio/mean": 1.0010185241699219, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3832540027797222, "clip_ratio/low_mean": 0.048370727337896824, "clip_ratio/low_min": 0.048370727337896824, "clip_ratio/high_mean": 0.0447253305464983, "clip_ratio/high_max": 0.0447253305464983, "clip_ratio/region_mean": 0.09309605788439512, "reward_total_mean": 0.8365767002105713, "reward_meter_mean": 0.8960349559783936, "reward_meter_std": 0.186055526137352, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9511100649833679, "reward_repeat_soft_std": 0.03043997846543789, "reward_judge_quality_mean": 0.6274999976158142, "reward_judge_quality_std": 0.2290196716785431, "reward_total_composite_mean": 0.8365767002105713, "reward_total_composite_std": 0.13514691591262817} {"timestamp_utc": "2026-04-13T05:13:34Z", "mode": "train", "global_step": 2738, "epoch": 0.2750376695128076, "loss": 0.0719, "grad_norm": 9.812220573425293, "learning_rate": 1.7060606060606062e-06, "num_tokens": 5102393.0, "completions/mean_length": 103.0, "completions/min_length": 80.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.0, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.9784235954284668, "rewards/meter/std": 0.009376212023198605, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7975363731384277, "rewards/repeat_soft/std": 0.09456966072320938, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.1524970829486847, "rewards/total_composite/mean": 0.785919189453125, "rewards/total_composite/std": 0.049928802996873856, "reward": 0.785919189453125, "reward_std": 0.04992878809571266, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08635202795267105, "sampling/sampling_logp_difference/max": 2.719801425933838, "sampling/importance_sampling_ratio/min": 0.30483099818229675, "sampling/importance_sampling_ratio/mean": 0.9998845458030701, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3423379361629486, "clip_ratio/low_mean": 0.02096265461295843, "clip_ratio/low_min": 0.02096265461295843, "clip_ratio/high_mean": 0.04816016275435686, "clip_ratio/high_max": 0.04816016275435686, "clip_ratio/region_mean": 0.06912281736731529, "reward_total_mean": 0.785919189453125, "reward_meter_mean": 0.9784235954284668, "reward_meter_std": 0.009376212023198605, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7975363731384277, "reward_repeat_soft_std": 0.09456966072320938, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.1524970829486847, "reward_total_composite_mean": 0.785919189453125, "reward_total_composite_std": 0.049928802996873856} {"timestamp_utc": "2026-04-13T05:13:41Z", "mode": "train", "global_step": 2739, "epoch": 0.2751381215469613, "loss": 0.0353, "grad_norm": 6.005005836486816, "learning_rate": 1.703030303030303e-06, "num_tokens": 5104625.0, "completions/mean_length": 113.0, "completions/min_length": 109.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.0, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.9894098043441772, "rewards/meter/std": 0.0019302217988297343, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8294169902801514, "rewards/repeat_soft/std": 0.07794439047574997, "rewards/judge_quality/mean": 0.30124998092651367, "rewards/judge_quality/std": 0.10398317128419876, "rewards/total_composite/mean": 0.7685511112213135, "rewards/total_composite/std": 0.035595644265413284, "reward": 0.7685511112213135, "reward_std": 0.035595621913671494, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.057356782257556915, "sampling/sampling_logp_difference/max": 1.551988124847412, "sampling/importance_sampling_ratio/min": 0.2118264138698578, "sampling/importance_sampling_ratio/mean": 1.0027544498443604, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.29036943800747395, "clip_ratio/low_mean": 0.021865422371774912, "clip_ratio/low_min": 0.021865422371774912, "clip_ratio/high_mean": 0.019311077427119017, "clip_ratio/high_max": 0.019311077427119017, "clip_ratio/region_mean": 0.04117649979889393, "reward_total_mean": 0.7685511112213135, "reward_meter_mean": 0.9894098043441772, "reward_meter_std": 0.0019302217988297343, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8294169902801514, "reward_repeat_soft_std": 0.07794439047574997, "reward_judge_quality_mean": 0.30124998092651367, "reward_judge_quality_std": 0.10398317128419876, "reward_total_composite_mean": 0.7685511112213135, "reward_total_composite_std": 0.035595644265413284} {"timestamp_utc": "2026-04-13T05:13:52Z", "mode": "train", "global_step": 2740, "epoch": 0.27523857358111503, "loss": -0.1283, "grad_norm": 3.1998958587646484, "learning_rate": 1.7000000000000002e-06, "num_tokens": 5106242.0, "completions/mean_length": 116.125, "completions/min_length": 55.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 59.57143020629883, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.5997518301010132, "rewards/meter/std": 0.41327863931655884, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9714324474334717, "rewards/repeat_soft/std": 0.020459050312638283, "rewards/judge_quality/mean": 0.8050000071525574, "rewards/judge_quality/std": 0.3252691328525543, "rewards/total_composite/mean": 0.7162919044494629, "rewards/total_composite/std": 0.336708664894104, "reward": 0.7162919044494629, "reward_std": 0.3367086350917816, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11459076404571533, "sampling/sampling_logp_difference/max": 1.4351706504821777, "sampling/importance_sampling_ratio/min": 0.3263993263244629, "sampling/importance_sampling_ratio/mean": 0.9904091358184814, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4704383797943592, "clip_ratio/low_mean": 0.02917477022856474, "clip_ratio/low_min": 0.02917477022856474, "clip_ratio/high_mean": 0.08026986010372639, "clip_ratio/high_max": 0.08026986010372639, "clip_ratio/region_mean": 0.10944463033229113, "reward_total_mean": 0.7162919044494629, "reward_meter_mean": 0.5997518301010132, "reward_meter_std": 0.41327863931655884, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9714324474334717, "reward_repeat_soft_std": 0.020459050312638283, "reward_judge_quality_mean": 0.8050000071525574, "reward_judge_quality_std": 0.3252691328525543, "reward_total_composite_mean": 0.7162919044494629, "reward_total_composite_std": 0.336708664894104} {"timestamp_utc": "2026-04-13T05:14:03Z", "mode": "train", "global_step": 2741, "epoch": 0.2753390256152687, "loss": -0.1124, "grad_norm": 1.6612247228622437, "learning_rate": 1.6969696969696973e-06, "num_tokens": 5107835.0, "completions/mean_length": 96.125, "completions/min_length": 33.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 36.71428680419922, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.8167057037353516, "rewards/meter/std": 0.3272828459739685, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9614260196685791, "rewards/repeat_soft/std": 0.03874724730849266, "rewards/judge_quality/mean": 0.6362500190734863, "rewards/judge_quality/std": 0.31459444761276245, "rewards/total_composite/mean": 0.7705260515213013, "rewards/total_composite/std": 0.32035762071609497, "reward": 0.7705260515213013, "reward_std": 0.32035762071609497, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09555637091398239, "sampling/sampling_logp_difference/max": 1.6799538135528564, "sampling/importance_sampling_ratio/min": 0.18638257682323456, "sampling/importance_sampling_ratio/mean": 1.0040374994277954, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.288998169824481, "clip_ratio/low_mean": 0.033386753872036934, "clip_ratio/low_min": 0.033386753872036934, "clip_ratio/high_mean": 0.06561987660825253, "clip_ratio/high_max": 0.06561987660825253, "clip_ratio/region_mean": 0.09900663048028946, "reward_total_mean": 0.7705260515213013, "reward_meter_mean": 0.8167057037353516, "reward_meter_std": 0.3272828459739685, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9614260196685791, "reward_repeat_soft_std": 0.03874724730849266, "reward_judge_quality_mean": 0.6362500190734863, "reward_judge_quality_std": 0.31459444761276245, "reward_total_composite_mean": 0.7705260515213013, "reward_total_composite_std": 0.32035762071609497} {"timestamp_utc": "2026-04-13T05:14:12Z", "mode": "train", "global_step": 2742, "epoch": 0.2754394776494224, "loss": -0.0305, "grad_norm": 4.535829544067383, "learning_rate": 1.6939393939393941e-06, "num_tokens": 5111253.0, "completions/mean_length": 204.25, "completions/min_length": 174.0, "completions/max_length": 227.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 204.25, "completions/min_terminated_length": 174.0, "completions/max_terminated_length": 227.0, "rewards/meter/mean": 0.9941853284835815, "rewards/meter/std": 0.002369991969317198, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7280257940292358, "rewards/repeat_soft/std": 0.10310235619544983, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7771859765052795, "rewards/total_composite/std": 0.024594461545348167, "reward": 0.7771859765052795, "reward_std": 0.02459445782005787, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.055015187710523605, "sampling/sampling_logp_difference/max": 2.362046480178833, "sampling/importance_sampling_ratio/min": 0.09422718733549118, "sampling/importance_sampling_ratio/mean": 1.0021345615386963, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.24050414375960827, "clip_ratio/low_mean": 0.021004737820476294, "clip_ratio/low_min": 0.021004737820476294, "clip_ratio/high_mean": 0.03403737582266331, "clip_ratio/high_max": 0.03403737582266331, "clip_ratio/region_mean": 0.0550421136431396, "reward_total_mean": 0.7771859765052795, "reward_meter_mean": 0.9941853284835815, "reward_meter_std": 0.002369991969317198, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7280257940292358, "reward_repeat_soft_std": 0.10310235619544983, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7771859765052795, "reward_total_composite_std": 0.024594461545348167} {"timestamp_utc": "2026-04-13T05:14:18Z", "mode": "train", "global_step": 2743, "epoch": 0.2755399296835761, "loss": -0.0144, "grad_norm": 7.607458591461182, "learning_rate": 1.6909090909090912e-06, "num_tokens": 5113100.0, "completions/mean_length": 49.875, "completions/min_length": 44.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.875, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.9951604604721069, "rewards/meter/std": 0.0015508191427215934, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9118542671203613, "rewards/repeat_soft/std": 0.029442893341183662, "rewards/judge_quality/mean": 0.6225000023841858, "rewards/judge_quality/std": 0.24656209349632263, "rewards/total_composite/mean": 0.8757575750350952, "rewards/total_composite/std": 0.07549859583377838, "reward": 0.8757575750350952, "reward_std": 0.07549858838319778, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07894125580787659, "sampling/sampling_logp_difference/max": 1.4797589778900146, "sampling/importance_sampling_ratio/min": 0.22769255936145782, "sampling/importance_sampling_ratio/mean": 0.9926438331604004, "sampling/importance_sampling_ratio/max": 1.777094841003418, "entropy": 0.3127066418528557, "clip_ratio/low_mean": 0.01701587811112404, "clip_ratio/low_min": 0.01701587811112404, "clip_ratio/high_mean": 0.028712607454508543, "clip_ratio/high_max": 0.028712607454508543, "clip_ratio/region_mean": 0.04572848556563258, "reward_total_mean": 0.8757575750350952, "reward_meter_mean": 0.9951604604721069, "reward_meter_std": 0.0015508191427215934, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9118542671203613, "reward_repeat_soft_std": 0.029442893341183662, "reward_judge_quality_mean": 0.6225000023841858, "reward_judge_quality_std": 0.24656209349632263, "reward_total_composite_mean": 0.8757575750350952, "reward_total_composite_std": 0.07549859583377838} {"timestamp_utc": "2026-04-13T05:14:25Z", "mode": "train", "global_step": 2744, "epoch": 0.2756403817177298, "loss": 0.0209, "grad_norm": 3.4097039699554443, "learning_rate": 1.687878787878788e-06, "num_tokens": 5115592.0, "completions/mean_length": 131.5, "completions/min_length": 123.0, "completions/max_length": 139.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.5, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.9715859293937683, "rewards/meter/std": 0.05791376158595085, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7139911651611328, "rewards/repeat_soft/std": 0.09265031665563583, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.8033628463745117, "rewards/total_composite/std": 0.062345314770936966, "reward": 0.8033628463745117, "reward_std": 0.06234531104564667, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05281757563352585, "sampling/sampling_logp_difference/max": 3.6844091415405273, "sampling/importance_sampling_ratio/min": 0.025112008675932884, "sampling/importance_sampling_ratio/mean": 1.0034140348434448, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.20785296894609928, "clip_ratio/low_mean": 0.027670096314977854, "clip_ratio/low_min": 0.027670096314977854, "clip_ratio/high_mean": 0.00675313058309257, "clip_ratio/high_max": 0.00675313058309257, "clip_ratio/region_mean": 0.034423226898070425, "reward_total_mean": 0.8033628463745117, "reward_meter_mean": 0.9715859293937683, "reward_meter_std": 0.05791376158595085, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7139911651611328, "reward_repeat_soft_std": 0.09265031665563583, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.8033628463745117, "reward_total_composite_std": 0.062345314770936966} {"timestamp_utc": "2026-04-13T05:14:32Z", "mode": "train", "global_step": 2745, "epoch": 0.27574083375188346, "loss": 0.0138, "grad_norm": 7.197882652282715, "learning_rate": 1.684848484848485e-06, "num_tokens": 5117254.0, "completions/mean_length": 57.75, "completions/min_length": 55.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.75, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9892073273658752, "rewards/meter/std": 0.004015904851257801, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9344595670700073, "rewards/repeat_soft/std": 0.05315641313791275, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.8082142472267151, "rewards/total_composite/std": 0.02289894036948681, "reward": 0.8082142472267151, "reward_std": 0.02289893850684166, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06264259666204453, "sampling/sampling_logp_difference/max": 0.9407896995544434, "sampling/importance_sampling_ratio/min": 0.39031949639320374, "sampling/importance_sampling_ratio/mean": 1.0111836194992065, "sampling/importance_sampling_ratio/max": 1.681404948234558, "entropy": 0.3322350010275841, "clip_ratio/low_mean": 0.0042372881434857845, "clip_ratio/low_min": 0.0042372881434857845, "clip_ratio/high_mean": 0.052057429449632764, "clip_ratio/high_max": 0.052057429449632764, "clip_ratio/region_mean": 0.05629471759311855, "reward_total_mean": 0.8082142472267151, "reward_meter_mean": 0.9892073273658752, "reward_meter_std": 0.004015904851257801, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9344595670700073, "reward_repeat_soft_std": 0.05315641313791275, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.8082142472267151, "reward_total_composite_std": 0.02289894036948681} {"timestamp_utc": "2026-04-13T05:14:39Z", "mode": "train", "global_step": 2746, "epoch": 0.27584128578603717, "loss": -0.0, "grad_norm": 7.988093852996826, "learning_rate": 1.6818181818181819e-06, "num_tokens": 5119359.0, "completions/mean_length": 95.125, "completions/min_length": 89.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.125, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.8431339263916016, "rewards/meter/std": 0.22682388126850128, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8813139796257019, "rewards/repeat_soft/std": 0.04976176097989082, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.762291669845581, "rewards/total_composite/std": 0.12410787492990494, "reward": 0.762291669845581, "reward_std": 0.12410786747932434, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08245062828063965, "sampling/sampling_logp_difference/max": 1.9493024349212646, "sampling/importance_sampling_ratio/min": 0.14237335324287415, "sampling/importance_sampling_ratio/mean": 1.00457763671875, "sampling/importance_sampling_ratio/max": 1.7795532941818237, "entropy": 0.38264819607138634, "clip_ratio/low_mean": 0.03417040826752782, "clip_ratio/low_min": 0.03417040826752782, "clip_ratio/high_mean": 0.05833917949348688, "clip_ratio/high_max": 0.05833917949348688, "clip_ratio/region_mean": 0.0925095877610147, "reward_total_mean": 0.762291669845581, "reward_meter_mean": 0.8431339263916016, "reward_meter_std": 0.22682388126850128, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8813139796257019, "reward_repeat_soft_std": 0.04976176097989082, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.762291669845581, "reward_total_composite_std": 0.12410787492990494} {"timestamp_utc": "2026-04-13T05:14:45Z", "mode": "train", "global_step": 2747, "epoch": 0.2759417378201909, "loss": -0.0225, "grad_norm": 7.070695877075195, "learning_rate": 1.678787878787879e-06, "num_tokens": 5121161.0, "completions/mean_length": 56.25, "completions/min_length": 50.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.25, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9578092098236084, "rewards/meter/std": 0.08628890663385391, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9245600700378418, "rewards/repeat_soft/std": 0.047399040311574936, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8005951046943665, "rewards/total_composite/std": 0.03928181529045105, "reward": 0.8005951046943665, "reward_std": 0.03928181156516075, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06195472925901413, "sampling/sampling_logp_difference/max": 1.6066513061523438, "sampling/importance_sampling_ratio/min": 0.20055809617042542, "sampling/importance_sampling_ratio/mean": 1.0085134506225586, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3479911796748638, "clip_ratio/low_mean": 0.01627192972227931, "clip_ratio/low_min": 0.01627192972227931, "clip_ratio/high_mean": 0.05973803298547864, "clip_ratio/high_max": 0.05973803298547864, "clip_ratio/region_mean": 0.07600996270775795, "reward_total_mean": 0.8005951046943665, "reward_meter_mean": 0.9578092098236084, "reward_meter_std": 0.08628890663385391, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9245600700378418, "reward_repeat_soft_std": 0.047399040311574936, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8005951046943665, "reward_total_composite_std": 0.03928181529045105} {"timestamp_utc": "2026-04-13T05:14:51Z", "mode": "train", "global_step": 2748, "epoch": 0.27604218985434453, "loss": 0.0393, "grad_norm": 10.906798362731934, "learning_rate": 1.675757575757576e-06, "num_tokens": 5122781.0, "completions/mean_length": 48.5, "completions/min_length": 40.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.5, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.7989785075187683, "rewards/meter/std": 0.2972187399864197, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9652544260025024, "rewards/repeat_soft/std": 0.03680083900690079, "rewards/judge_quality/mean": 0.65625, "rewards/judge_quality/std": 0.14647647738456726, "rewards/total_composite/mean": 0.8029407262802124, "rewards/total_composite/std": 0.1292758733034134, "reward": 0.8029407262802124, "reward_std": 0.1292758733034134, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11062220484018326, "sampling/sampling_logp_difference/max": 2.3792638778686523, "sampling/importance_sampling_ratio/min": 0.09261873364448547, "sampling/importance_sampling_ratio/mean": 0.9971563220024109, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3864385262131691, "clip_ratio/low_mean": 0.03505896218121052, "clip_ratio/low_min": 0.03505896218121052, "clip_ratio/high_mean": 0.06997093046084046, "clip_ratio/high_max": 0.06997093046084046, "clip_ratio/region_mean": 0.10502989264205098, "reward_total_mean": 0.8029407262802124, "reward_meter_mean": 0.7989785075187683, "reward_meter_std": 0.2972187399864197, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9652544260025024, "reward_repeat_soft_std": 0.03680083900690079, "reward_judge_quality_mean": 0.65625, "reward_judge_quality_std": 0.14647647738456726, "reward_total_composite_mean": 0.8029407262802124, "reward_total_composite_std": 0.1292758733034134} {"timestamp_utc": "2026-04-13T05:14:58Z", "mode": "train", "global_step": 2749, "epoch": 0.27614264188849824, "loss": 0.0273, "grad_norm": 5.807712554931641, "learning_rate": 1.6727272727272728e-06, "num_tokens": 5124849.0, "completions/mean_length": 86.5, "completions/min_length": 82.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.5, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9925795793533325, "rewards/meter/std": 0.00183397950604558, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8302425146102905, "rewards/repeat_soft/std": 0.08386033028364182, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8056850433349609, "rewards/total_composite/std": 0.009137063287198544, "reward": 0.8056850433349609, "reward_std": 0.009137062355875969, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.056007616221904755, "sampling/sampling_logp_difference/max": 1.1991662979125977, "sampling/importance_sampling_ratio/min": 0.3014453947544098, "sampling/importance_sampling_ratio/mean": 1.0042564868927002, "sampling/importance_sampling_ratio/max": 1.8329919576644897, "entropy": 0.26944844238460064, "clip_ratio/low_mean": 0.024166989605873823, "clip_ratio/low_min": 0.024166989605873823, "clip_ratio/high_mean": 0.030639416771009564, "clip_ratio/high_max": 0.030639416771009564, "clip_ratio/region_mean": 0.05480640637688339, "reward_total_mean": 0.8056850433349609, "reward_meter_mean": 0.9925795793533325, "reward_meter_std": 0.00183397950604558, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8302425146102905, "reward_repeat_soft_std": 0.08386033028364182, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8056850433349609, "reward_total_composite_std": 0.009137063287198544} {"timestamp_utc": "2026-04-13T05:15:04Z", "mode": "train", "global_step": 2750, "epoch": 0.27624309392265195, "loss": 0.0263, "grad_norm": 11.320253372192383, "learning_rate": 1.6696969696969698e-06, "num_tokens": 5126815.0, "completions/mean_length": 79.75, "completions/min_length": 74.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.75, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.9888013005256653, "rewards/meter/std": 0.0021845349110662937, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8339663147926331, "rewards/repeat_soft/std": 0.06074106693267822, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.0975411981344223, "rewards/total_composite/mean": 0.7953572273254395, "rewards/total_composite/std": 0.026251213625073433, "reward": 0.7953572273254395, "reward_std": 0.026251211762428284, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09872998297214508, "sampling/sampling_logp_difference/max": 2.383737564086914, "sampling/importance_sampling_ratio/min": 0.09220530837774277, "sampling/importance_sampling_ratio/mean": 1.0027340650558472, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.393969539552927, "clip_ratio/low_mean": 0.02173996903002262, "clip_ratio/low_min": 0.02173996903002262, "clip_ratio/high_mean": 0.06589851574972272, "clip_ratio/high_max": 0.06589851574972272, "clip_ratio/region_mean": 0.08763848477974534, "reward_total_mean": 0.7953572273254395, "reward_meter_mean": 0.9888013005256653, "reward_meter_std": 0.0021845349110662937, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8339663147926331, "reward_repeat_soft_std": 0.06074106693267822, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.0975411981344223, "reward_total_composite_mean": 0.7953572273254395, "reward_total_composite_std": 0.026251213625073433} {"timestamp_utc": "2026-04-13T05:15:59Z", "mode": "eval", "global_step": 2750, "epoch": 0.27624309392265195, "eval_loss": NaN, "eval_runtime": 54.1076, "eval_samples_per_second": 1.479, "eval_steps_per_second": 0.185, "eval_num_tokens": 5126815.0, "eval_completions/mean_length": 104.1625, "eval_completions/min_length": 41.3, "eval_completions/max_length": 226.6, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 94.08392944335938, "eval_completions/min_terminated_length": 41.3, "eval_completions/max_terminated_length": 160.0, "eval_rewards/meter/mean": 0.90377357006073, "eval_rewards/meter/std": 0.16382857505232096, "eval_rewards/count_adherence/mean": 0.979583328962326, "eval_rewards/count_adherence/std": 0.05286311954259872, "eval_rewards/hard_gate/mean": 0.9875, "eval_rewards/hard_gate/std": 0.03535533845424652, "eval_rewards/repeat_soft/mean": 0.8961685299873352, "eval_rewards/repeat_soft/std": 0.09489056058228015, "eval_rewards/judge_quality/mean": 0.42487499713897703, "eval_rewards/judge_quality/std": 0.13789428481832147, "eval_rewards/total_composite/mean": 0.7641787707805634, "eval_rewards/total_composite/std": 0.10884297974407672, "eval_reward": 0.7641787707805634, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.03867025170475245, "eval_sampling/sampling_logp_difference/max": 0.947283935546875, "eval_sampling/importance_sampling_ratio/min": 0.39218426048755645, "eval_sampling/importance_sampling_ratio/mean": 1.0074607491493226, "eval_sampling/importance_sampling_ratio/max": 1.3516329050064086, "eval_entropy": 0.37834500074386596, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7641787707805634, "eval_reward_meter_mean": 0.90377357006073, "eval_reward_meter_std": 0.16382857505232096, "eval_reward_count_adherence_mean": 0.979583328962326, "eval_reward_count_adherence_std": 0.05286311954259872, "eval_reward_hard_gate_mean": 0.9875, "eval_reward_hard_gate_std": 0.03535533845424652, "eval_reward_repeat_soft_mean": 0.8961685299873352, "eval_reward_repeat_soft_std": 0.09489056058228015, "eval_reward_judge_quality_mean": 0.42487499713897703, "eval_reward_judge_quality_std": 0.13789428481832147, "eval_reward_total_composite_mean": 0.7641787707805634, "eval_reward_total_composite_std": 0.10884297974407672} {"timestamp_utc": "2026-04-13T05:16:08Z", "mode": "train", "global_step": 2751, "epoch": 0.2763435459568056, "loss": -0.0271, "grad_norm": 14.57835865020752, "learning_rate": 1.6666666666666667e-06, "num_tokens": 5128145.0, "completions/mean_length": 22.25, "completions/min_length": 21.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.25, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.9806317090988159, "rewards/meter/std": 0.01931656338274479, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9413700103759766, "rewards/repeat_soft/std": 0.040233537554740906, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.8324212431907654, "rewards/total_composite/std": 0.05529043823480606, "reward": 0.8324212431907654, "reward_std": 0.05529043823480606, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07681550085544586, "sampling/sampling_logp_difference/max": 3.154601573944092, "sampling/importance_sampling_ratio/min": 0.04265538975596428, "sampling/importance_sampling_ratio/mean": 1.005128026008606, "sampling/importance_sampling_ratio/max": 1.5377445220947266, "entropy": 0.3586646690964699, "clip_ratio/low_mean": 0.09569687815383077, "clip_ratio/low_min": 0.09569687815383077, "clip_ratio/high_mean": 0.004999999888241291, "clip_ratio/high_max": 0.004999999888241291, "clip_ratio/region_mean": 0.10069687804207206, "reward_total_mean": 0.8324212431907654, "reward_meter_mean": 0.9806317090988159, "reward_meter_std": 0.01931656338274479, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9413700103759766, "reward_repeat_soft_std": 0.040233537554740906, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.8324212431907654, "reward_total_composite_std": 0.05529043823480606} {"timestamp_utc": "2026-04-13T05:16:15Z", "mode": "train", "global_step": 2752, "epoch": 0.2764439979909593, "loss": 0.0557, "grad_norm": 9.994974136352539, "learning_rate": 1.6636363636363637e-06, "num_tokens": 5130159.0, "completions/mean_length": 79.75, "completions/min_length": 73.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.75, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.935316801071167, "rewards/meter/std": 0.02231025882065296, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9371764659881592, "rewards/repeat_soft/std": 0.03883935511112213, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.8206102252006531, "rewards/total_composite/std": 0.062416478991508484, "reward": 0.8206102252006531, "reward_std": 0.062416478991508484, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09214691817760468, "sampling/sampling_logp_difference/max": 1.391026496887207, "sampling/importance_sampling_ratio/min": 0.2488197684288025, "sampling/importance_sampling_ratio/mean": 1.0096420049667358, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37552786991000175, "clip_ratio/low_mean": 0.06896527856588364, "clip_ratio/low_min": 0.06896527856588364, "clip_ratio/high_mean": 0.020142393186688423, "clip_ratio/high_max": 0.020142393186688423, "clip_ratio/region_mean": 0.08910767175257206, "reward_total_mean": 0.8206102252006531, "reward_meter_mean": 0.935316801071167, "reward_meter_std": 0.02231025882065296, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9371764659881592, "reward_repeat_soft_std": 0.03883935511112213, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.8206102252006531, "reward_total_composite_std": 0.062416478991508484} {"timestamp_utc": "2026-04-13T05:16:27Z", "mode": "train", "global_step": 2753, "epoch": 0.276544450025113, "loss": -0.1592, "grad_norm": 1.4679391384124756, "learning_rate": 1.6606060606060605e-06, "num_tokens": 5131803.0, "completions/mean_length": 119.5, "completions/min_length": 59.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 63.42857360839844, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9877691268920898, "rewards/meter/std": 0.009354579262435436, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8422646522521973, "rewards/repeat_soft/std": 0.11089050769805908, "rewards/judge_quality/mean": 0.44875001907348633, "rewards/judge_quality/std": 0.2105392962694168, "rewards/total_composite/mean": 0.7258384823799133, "rewards/total_composite/std": 0.2969808280467987, "reward": 0.7258384823799133, "reward_std": 0.2969808280467987, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0716036707162857, "sampling/sampling_logp_difference/max": 1.3517868518829346, "sampling/importance_sampling_ratio/min": 0.2587774395942688, "sampling/importance_sampling_ratio/mean": 1.009050726890564, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36096760630607605, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.07118743704631925, "clip_ratio/high_max": 0.07118743704631925, "clip_ratio/region_mean": 0.07118743704631925, "reward_total_mean": 0.7258384823799133, "reward_meter_mean": 0.9877691268920898, "reward_meter_std": 0.009354579262435436, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8422646522521973, "reward_repeat_soft_std": 0.11089050769805908, "reward_judge_quality_mean": 0.44875001907348633, "reward_judge_quality_std": 0.2105392962694168, "reward_total_composite_mean": 0.7258384823799133, "reward_total_composite_std": 0.2969808280467987} {"timestamp_utc": "2026-04-13T05:16:34Z", "mode": "train", "global_step": 2754, "epoch": 0.2766449020592667, "loss": 0.012, "grad_norm": 7.115744113922119, "learning_rate": 1.6575757575757578e-06, "num_tokens": 5133772.0, "completions/mean_length": 69.125, "completions/min_length": 61.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.125, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9881517887115479, "rewards/meter/std": 0.00809910986572504, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9454888105392456, "rewards/repeat_soft/std": 0.053881656378507614, "rewards/judge_quality/mean": 0.6274999976158142, "rewards/judge_quality/std": 0.21952873468399048, "rewards/total_composite/mean": 0.877467155456543, "rewards/total_composite/std": 0.0675932988524437, "reward": 0.877467155456543, "reward_std": 0.0675932914018631, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1034553200006485, "sampling/sampling_logp_difference/max": 2.1722843647003174, "sampling/importance_sampling_ratio/min": 0.11391708999872208, "sampling/importance_sampling_ratio/mean": 1.009818196296692, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.544053602963686, "clip_ratio/low_mean": 0.05098322872072458, "clip_ratio/low_min": 0.05098322872072458, "clip_ratio/high_mean": 0.05322598107159138, "clip_ratio/high_max": 0.05322598107159138, "clip_ratio/region_mean": 0.10420920979231596, "reward_total_mean": 0.877467155456543, "reward_meter_mean": 0.9881517887115479, "reward_meter_std": 0.00809910986572504, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9454888105392456, "reward_repeat_soft_std": 0.053881656378507614, "reward_judge_quality_mean": 0.6274999976158142, "reward_judge_quality_std": 0.21952873468399048, "reward_total_composite_mean": 0.877467155456543, "reward_total_composite_std": 0.0675932988524437} {"timestamp_utc": "2026-04-13T05:16:41Z", "mode": "train", "global_step": 2755, "epoch": 0.2767453540934204, "loss": -0.0132, "grad_norm": 9.614466667175293, "learning_rate": 1.6545454545454548e-06, "num_tokens": 5135910.0, "completions/mean_length": 87.25, "completions/min_length": 79.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.25, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.9904704689979553, "rewards/meter/std": 0.0048414976336061954, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9621832370758057, "rewards/repeat_soft/std": 0.019995922222733498, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.811555027961731, "rewards/total_composite/std": 0.017405180260539055, "reward": 0.811555027961731, "reward_std": 0.017405183985829353, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1005145013332367, "sampling/sampling_logp_difference/max": 2.4980921745300293, "sampling/importance_sampling_ratio/min": 0.08224175870418549, "sampling/importance_sampling_ratio/mean": 1.0072044134140015, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4438781812787056, "clip_ratio/low_mean": 0.007911392487585545, "clip_ratio/low_min": 0.007911392487585545, "clip_ratio/high_mean": 0.0676797847263515, "clip_ratio/high_max": 0.0676797847263515, "clip_ratio/region_mean": 0.07559117721393704, "reward_total_mean": 0.811555027961731, "reward_meter_mean": 0.9904704689979553, "reward_meter_std": 0.0048414976336061954, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9621832370758057, "reward_repeat_soft_std": 0.019995922222733498, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.811555027961731, "reward_total_composite_std": 0.017405180260539055} {"timestamp_utc": "2026-04-13T05:16:49Z", "mode": "train", "global_step": 2756, "epoch": 0.2768458061275741, "loss": -0.039, "grad_norm": 5.635105609893799, "learning_rate": 1.6515151515151517e-06, "num_tokens": 5138963.0, "completions/mean_length": 195.625, "completions/min_length": 166.0, "completions/max_length": 219.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 195.625, "completions/min_terminated_length": 166.0, "completions/max_terminated_length": 219.0, "rewards/meter/mean": 0.9928933382034302, "rewards/meter/std": 0.0028256357181817293, "rewards/count_adherence/mean": 0.9166666269302368, "rewards/count_adherence/std": 0.0890870913863182, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9227234125137329, "rewards/repeat_soft/std": 0.019903438165783882, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8025743961334229, "rewards/total_composite/std": 0.0126614635810256, "reward": 0.8025743961334229, "reward_std": 0.01266145333647728, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08949941396713257, "sampling/sampling_logp_difference/max": 2.9321742057800293, "sampling/importance_sampling_ratio/min": 0.05328106880187988, "sampling/importance_sampling_ratio/mean": 1.0047333240509033, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4030468501150608, "clip_ratio/low_mean": 0.039001185446977615, "clip_ratio/low_min": 0.039001185446977615, "clip_ratio/high_mean": 0.04114541132003069, "clip_ratio/high_max": 0.04114541132003069, "clip_ratio/region_mean": 0.0801465967670083, "reward_total_mean": 0.8025743961334229, "reward_meter_mean": 0.9928933382034302, "reward_meter_std": 0.0028256357181817293, "reward_count_adherence_mean": 0.9166666269302368, "reward_count_adherence_std": 0.0890870913863182, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9227234125137329, "reward_repeat_soft_std": 0.019903438165783882, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8025743961334229, "reward_total_composite_std": 0.0126614635810256} {"timestamp_utc": "2026-04-13T05:17:01Z", "mode": "train", "global_step": 2757, "epoch": 0.2769462581617278, "loss": -0.1211, "grad_norm": 1.59152090549469, "learning_rate": 1.6484848484848487e-06, "num_tokens": 5140399.0, "completions/mean_length": 100.5, "completions/min_length": 40.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 41.71428680419922, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.7653188705444336, "rewards/meter/std": 0.329611212015152, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9134917259216309, "rewards/repeat_soft/std": 0.07463657110929489, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.28965190052986145, "rewards/total_composite/mean": 0.7022426724433899, "rewards/total_composite/std": 0.30139708518981934, "reward": 0.7022426724433899, "reward_std": 0.30139708518981934, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06632012873888016, "sampling/sampling_logp_difference/max": 0.7830343246459961, "sampling/importance_sampling_ratio/min": 0.4570171535015106, "sampling/importance_sampling_ratio/mean": 1.0127583742141724, "sampling/importance_sampling_ratio/max": 1.8066060543060303, "entropy": 0.2854253277182579, "clip_ratio/low_mean": 0.011627906933426857, "clip_ratio/low_min": 0.011627906933426857, "clip_ratio/high_mean": 0.054560762364417315, "clip_ratio/high_max": 0.054560762364417315, "clip_ratio/region_mean": 0.06618866929784417, "reward_total_mean": 0.7022426724433899, "reward_meter_mean": 0.7653188705444336, "reward_meter_std": 0.329611212015152, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9134917259216309, "reward_repeat_soft_std": 0.07463657110929489, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.28965190052986145, "reward_total_composite_mean": 0.7022426724433899, "reward_total_composite_std": 0.30139708518981934} {"timestamp_utc": "2026-04-13T05:17:08Z", "mode": "train", "global_step": 2758, "epoch": 0.27704671019588145, "loss": 0.018, "grad_norm": 7.672840118408203, "learning_rate": 1.6454545454545455e-06, "num_tokens": 5142228.0, "completions/mean_length": 64.625, "completions/min_length": 63.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.625, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.964657187461853, "rewards/meter/std": 0.041993025690317154, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9092907905578613, "rewards/repeat_soft/std": 0.059894949197769165, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7946498394012451, "rewards/total_composite/std": 0.023497503250837326, "reward": 0.7946498394012451, "reward_std": 0.023497486487030983, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07625051587820053, "sampling/sampling_logp_difference/max": 1.8410606384277344, "sampling/importance_sampling_ratio/min": 0.15864907205104828, "sampling/importance_sampling_ratio/mean": 1.0190197229385376, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3726390413939953, "clip_ratio/low_mean": 0.03846153849735856, "clip_ratio/low_min": 0.03846153849735856, "clip_ratio/high_mean": 0.05246966611593962, "clip_ratio/high_max": 0.05246966611593962, "clip_ratio/region_mean": 0.09093120461329818, "reward_total_mean": 0.7946498394012451, "reward_meter_mean": 0.964657187461853, "reward_meter_std": 0.041993025690317154, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9092907905578613, "reward_repeat_soft_std": 0.059894949197769165, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7946498394012451, "reward_total_composite_std": 0.023497503250837326} {"timestamp_utc": "2026-04-13T05:17:14Z", "mode": "train", "global_step": 2759, "epoch": 0.27714716223003516, "loss": 0.0446, "grad_norm": 7.825084686279297, "learning_rate": 1.6424242424242426e-06, "num_tokens": 5143928.0, "completions/mean_length": 65.5, "completions/min_length": 56.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.5, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.995009183883667, "rewards/meter/std": 0.0020547485910356045, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9592829346656799, "rewards/repeat_soft/std": 0.04751931503415108, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.8594323992729187, "rewards/total_composite/std": 0.06985511630773544, "reward": 0.8594323992729187, "reward_std": 0.06985511630773544, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09271345287561417, "sampling/sampling_logp_difference/max": 1.7344622611999512, "sampling/importance_sampling_ratio/min": 0.17649509012699127, "sampling/importance_sampling_ratio/mean": 0.9864488244056702, "sampling/importance_sampling_ratio/max": 1.8752834796905518, "entropy": 0.41930732131004333, "clip_ratio/low_mean": 0.03979918081313372, "clip_ratio/low_min": 0.03979918081313372, "clip_ratio/high_mean": 0.01753421314060688, "clip_ratio/high_max": 0.01753421314060688, "clip_ratio/region_mean": 0.0573333939537406, "reward_total_mean": 0.8594323992729187, "reward_meter_mean": 0.995009183883667, "reward_meter_std": 0.0020547485910356045, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9592829346656799, "reward_repeat_soft_std": 0.04751931503415108, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.8594323992729187, "reward_total_composite_std": 0.06985511630773544} {"timestamp_utc": "2026-04-13T05:17:21Z", "mode": "train", "global_step": 2760, "epoch": 0.27724761426418887, "loss": -0.0016, "grad_norm": 6.858184814453125, "learning_rate": 1.6393939393939396e-06, "num_tokens": 5145914.0, "completions/mean_length": 85.25, "completions/min_length": 79.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.25, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.9791229963302612, "rewards/meter/std": 0.012721416540443897, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8518903851509094, "rewards/repeat_soft/std": 0.0717448890209198, "rewards/judge_quality/mean": 0.35999998450279236, "rewards/judge_quality/std": 0.09165150672197342, "rewards/total_composite/mean": 0.7837944030761719, "rewards/total_composite/std": 0.03383592516183853, "reward": 0.7837944030761719, "reward_std": 0.03383592888712883, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0760217234492302, "sampling/sampling_logp_difference/max": 2.491121530532837, "sampling/importance_sampling_ratio/min": 0.08281703293323517, "sampling/importance_sampling_ratio/mean": 0.9995399713516235, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3298765569925308, "clip_ratio/low_mean": 0.021264893002808094, "clip_ratio/low_min": 0.021264893002808094, "clip_ratio/high_mean": 0.044450194109231234, "clip_ratio/high_max": 0.044450194109231234, "clip_ratio/region_mean": 0.06571508711203933, "reward_total_mean": 0.7837944030761719, "reward_meter_mean": 0.9791229963302612, "reward_meter_std": 0.012721416540443897, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8518903851509094, "reward_repeat_soft_std": 0.0717448890209198, "reward_judge_quality_mean": 0.35999998450279236, "reward_judge_quality_std": 0.09165150672197342, "reward_total_composite_mean": 0.7837944030761719, "reward_total_composite_std": 0.03383592516183853} {"timestamp_utc": "2026-04-13T05:17:29Z", "mode": "train", "global_step": 2761, "epoch": 0.2773480662983425, "loss": 0.0413, "grad_norm": 5.05848503112793, "learning_rate": 1.6363636363636365e-06, "num_tokens": 5148642.0, "completions/mean_length": 141.0, "completions/min_length": 131.0, "completions/max_length": 161.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 141.0, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 161.0, "rewards/meter/mean": 0.951926589012146, "rewards/meter/std": 0.06130475550889969, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9033810496330261, "rewards/repeat_soft/std": 0.0385141558945179, "rewards/judge_quality/mean": 0.3050000071525574, "rewards/judge_quality/std": 0.07892129570245743, "rewards/total_composite/mean": 0.7602050304412842, "rewards/total_composite/std": 0.03212323784828186, "reward": 0.7602050304412842, "reward_std": 0.03212323784828186, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07067319005727768, "sampling/sampling_logp_difference/max": 2.3392744064331055, "sampling/importance_sampling_ratio/min": 0.09639756381511688, "sampling/importance_sampling_ratio/mean": 1.0041285753250122, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.32554028183221817, "clip_ratio/low_mean": 0.022779884049668908, "clip_ratio/low_min": 0.022779884049668908, "clip_ratio/high_mean": 0.04215629585087299, "clip_ratio/high_max": 0.04215629585087299, "clip_ratio/region_mean": 0.0649361799005419, "reward_total_mean": 0.7602050304412842, "reward_meter_mean": 0.951926589012146, "reward_meter_std": 0.06130475550889969, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9033810496330261, "reward_repeat_soft_std": 0.0385141558945179, "reward_judge_quality_mean": 0.3050000071525574, "reward_judge_quality_std": 0.07892129570245743, "reward_total_composite_mean": 0.7602050304412842, "reward_total_composite_std": 0.03212323784828186} {"timestamp_utc": "2026-04-13T05:17:35Z", "mode": "train", "global_step": 2762, "epoch": 0.27744851833249623, "loss": 0.0356, "grad_norm": 8.394858360290527, "learning_rate": 1.6333333333333335e-06, "num_tokens": 5150530.0, "completions/mean_length": 74.0, "completions/min_length": 64.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.981909990310669, "rewards/meter/std": 0.007504105567932129, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9517320990562439, "rewards/repeat_soft/std": 0.03770388290286064, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.8505327105522156, "rewards/total_composite/std": 0.07104675471782684, "reward": 0.8505327105522156, "reward_std": 0.07104675471782684, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10413072258234024, "sampling/sampling_logp_difference/max": 1.995124101638794, "sampling/importance_sampling_ratio/min": 0.1359967738389969, "sampling/importance_sampling_ratio/mean": 0.998586118221283, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4510823115706444, "clip_ratio/low_mean": 0.055316624231636524, "clip_ratio/low_min": 0.055316624231636524, "clip_ratio/high_mean": 0.030357143841683865, "clip_ratio/high_max": 0.030357143841683865, "clip_ratio/region_mean": 0.08567376807332039, "reward_total_mean": 0.8505327105522156, "reward_meter_mean": 0.981909990310669, "reward_meter_std": 0.007504105567932129, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9517320990562439, "reward_repeat_soft_std": 0.03770388290286064, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.8505327105522156, "reward_total_composite_std": 0.07104675471782684} {"timestamp_utc": "2026-04-13T05:17:42Z", "mode": "train", "global_step": 2763, "epoch": 0.27754897036664994, "loss": 0.047, "grad_norm": 6.178488731384277, "learning_rate": 1.6303030303030303e-06, "num_tokens": 5152884.0, "completions/mean_length": 92.25, "completions/min_length": 83.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.25, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.9884270429611206, "rewards/meter/std": 0.006547450087964535, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7309011220932007, "rewards/repeat_soft/std": 0.11093078553676605, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.7646322846412659, "rewards/total_composite/std": 0.032714542001485825, "reward": 0.7646322846412659, "reward_std": 0.03271452337503433, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0635526105761528, "sampling/sampling_logp_difference/max": 1.1017465591430664, "sampling/importance_sampling_ratio/min": 0.3322902023792267, "sampling/importance_sampling_ratio/mean": 1.0003761053085327, "sampling/importance_sampling_ratio/max": 1.8691993951797485, "entropy": 0.2844458017498255, "clip_ratio/low_mean": 0.026392717380076647, "clip_ratio/low_min": 0.026392717380076647, "clip_ratio/high_mean": 0.031182160135358572, "clip_ratio/high_max": 0.031182160135358572, "clip_ratio/region_mean": 0.05757487751543522, "reward_total_mean": 0.7646322846412659, "reward_meter_mean": 0.9884270429611206, "reward_meter_std": 0.006547450087964535, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7309011220932007, "reward_repeat_soft_std": 0.11093078553676605, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.7646322846412659, "reward_total_composite_std": 0.032714542001485825} {"timestamp_utc": "2026-04-13T05:17:53Z", "mode": "train", "global_step": 2764, "epoch": 0.2776494224008036, "loss": -0.1352, "grad_norm": 2.0849242210388184, "learning_rate": 1.6272727272727274e-06, "num_tokens": 5154640.0, "completions/mean_length": 111.5, "completions/min_length": 52.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 54.28571701049805, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.7640033960342407, "rewards/meter/std": 0.3492111563682556, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9835054874420166, "rewards/repeat_soft/std": 0.014770859852433205, "rewards/judge_quality/mean": 0.4300000071525574, "rewards/judge_quality/std": 0.18150167167186737, "rewards/total_composite/mean": 0.6880271434783936, "rewards/total_composite/std": 0.2909415662288666, "reward": 0.6880271434783936, "reward_std": 0.2909415662288666, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09803500026464462, "sampling/sampling_logp_difference/max": 1.4975345134735107, "sampling/importance_sampling_ratio/min": 0.22368097305297852, "sampling/importance_sampling_ratio/mean": 1.0073519945144653, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5678685717284679, "clip_ratio/low_mean": 0.024357769638299942, "clip_ratio/low_min": 0.024357769638299942, "clip_ratio/high_mean": 0.0724860792979598, "clip_ratio/high_max": 0.0724860792979598, "clip_ratio/region_mean": 0.09684384893625975, "reward_total_mean": 0.6880271434783936, "reward_meter_mean": 0.7640033960342407, "reward_meter_std": 0.3492111563682556, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9835054874420166, "reward_repeat_soft_std": 0.014770859852433205, "reward_judge_quality_mean": 0.4300000071525574, "reward_judge_quality_std": 0.18150167167186737, "reward_total_composite_mean": 0.6880271434783936, "reward_total_composite_std": 0.2909415662288666} {"timestamp_utc": "2026-04-13T05:18:01Z", "mode": "train", "global_step": 2765, "epoch": 0.2777498744349573, "loss": -0.0258, "grad_norm": 7.317401885986328, "learning_rate": 1.6242424242424242e-06, "num_tokens": 5157022.0, "completions/mean_length": 112.75, "completions/min_length": 102.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.75, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.9755488634109497, "rewards/meter/std": 0.011888561770319939, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.924988865852356, "rewards/repeat_soft/std": 0.03418053686618805, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.8011208772659302, "rewards/total_composite/std": 0.017849387601017952, "reward": 0.8011208772659302, "reward_std": 0.017849382013082504, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09075883775949478, "sampling/sampling_logp_difference/max": 1.9381694793701172, "sampling/importance_sampling_ratio/min": 0.21090181171894073, "sampling/importance_sampling_ratio/mean": 0.9982609748840332, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4111527167260647, "clip_ratio/low_mean": 0.03144313953816891, "clip_ratio/low_min": 0.03144313953816891, "clip_ratio/high_mean": 0.05968761537224054, "clip_ratio/high_max": 0.05968761537224054, "clip_ratio/region_mean": 0.09113075491040945, "reward_total_mean": 0.8011208772659302, "reward_meter_mean": 0.9755488634109497, "reward_meter_std": 0.011888561770319939, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.924988865852356, "reward_repeat_soft_std": 0.03418053686618805, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.8011208772659302, "reward_total_composite_std": 0.017849387601017952} {"timestamp_utc": "2026-04-13T05:18:07Z", "mode": "train", "global_step": 2766, "epoch": 0.277850326469111, "loss": -0.0164, "grad_norm": 8.057735443115234, "learning_rate": 1.6212121212121213e-06, "num_tokens": 5158698.0, "completions/mean_length": 53.5, "completions/min_length": 42.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.5, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9519268274307251, "rewards/meter/std": 0.07198746502399445, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8860430717468262, "rewards/repeat_soft/std": 0.09008540213108063, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.8128463625907898, "rewards/total_composite/std": 0.02643771655857563, "reward": 0.8128463625907898, "reward_std": 0.026437710970640182, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07392778992652893, "sampling/sampling_logp_difference/max": 1.0648982524871826, "sampling/importance_sampling_ratio/min": 0.3447629511356354, "sampling/importance_sampling_ratio/mean": 1.0020865201950073, "sampling/importance_sampling_ratio/max": 1.7704235315322876, "entropy": 0.4113520532846451, "clip_ratio/low_mean": 0.02678711572661996, "clip_ratio/low_min": 0.02678711572661996, "clip_ratio/high_mean": 0.02456761058419943, "clip_ratio/high_max": 0.02456761058419943, "clip_ratio/region_mean": 0.05135472631081939, "reward_total_mean": 0.8128463625907898, "reward_meter_mean": 0.9519268274307251, "reward_meter_std": 0.07198746502399445, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8860430717468262, "reward_repeat_soft_std": 0.09008540213108063, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.8128463625907898, "reward_total_composite_std": 0.02643771655857563} {"timestamp_utc": "2026-04-13T05:18:14Z", "mode": "train", "global_step": 2767, "epoch": 0.2779507785032647, "loss": 0.0158, "grad_norm": 9.038031578063965, "learning_rate": 1.618181818181818e-06, "num_tokens": 5160431.0, "completions/mean_length": 50.625, "completions/min_length": 47.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.625, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9827389717102051, "rewards/meter/std": 0.02452097274363041, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9918825030326843, "rewards/repeat_soft/std": 0.009393032640218735, "rewards/judge_quality/mean": 0.6575000286102295, "rewards/judge_quality/std": 0.1505940854549408, "rewards/total_composite/mean": 0.8886707425117493, "rewards/total_composite/std": 0.04338680952787399, "reward": 0.8886707425117493, "reward_std": 0.04338681325316429, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08978890627622604, "sampling/sampling_logp_difference/max": 1.3122878074645996, "sampling/importance_sampling_ratio/min": 0.3436375856399536, "sampling/importance_sampling_ratio/mean": 1.0096170902252197, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40238820388913155, "clip_ratio/low_mean": 0.03975490154698491, "clip_ratio/low_min": 0.03975490154698491, "clip_ratio/high_mean": 0.0385049763135612, "clip_ratio/high_max": 0.0385049763135612, "clip_ratio/region_mean": 0.07825987786054611, "reward_total_mean": 0.8886707425117493, "reward_meter_mean": 0.9827389717102051, "reward_meter_std": 0.02452097274363041, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9918825030326843, "reward_repeat_soft_std": 0.009393032640218735, "reward_judge_quality_mean": 0.6575000286102295, "reward_judge_quality_std": 0.1505940854549408, "reward_total_composite_mean": 0.8886707425117493, "reward_total_composite_std": 0.04338680952787399} {"timestamp_utc": "2026-04-13T05:18:20Z", "mode": "train", "global_step": 2768, "epoch": 0.27805123053741837, "loss": 0.0314, "grad_norm": 8.422323226928711, "learning_rate": 1.6151515151515153e-06, "num_tokens": 5161819.0, "completions/mean_length": 31.5, "completions/min_length": 28.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.5, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9477019309997559, "rewards/meter/std": 0.08215654641389847, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.948753833770752, "rewards/repeat_soft/std": 0.035154614597558975, "rewards/judge_quality/mean": 0.44624999165534973, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8052163124084473, "rewards/total_composite/std": 0.03599754348397255, "reward": 0.8052163124084473, "reward_std": 0.03599753975868225, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08037245273590088, "sampling/sampling_logp_difference/max": 1.708925724029541, "sampling/importance_sampling_ratio/min": 0.18106019496917725, "sampling/importance_sampling_ratio/mean": 1.006343126296997, "sampling/importance_sampling_ratio/max": 1.951904535293579, "entropy": 0.351595651358366, "clip_ratio/low_mean": 0.01515151560306549, "clip_ratio/low_min": 0.01515151560306549, "clip_ratio/high_mean": 0.07026847545057535, "clip_ratio/high_max": 0.07026847545057535, "clip_ratio/region_mean": 0.08541999105364084, "reward_total_mean": 0.8052163124084473, "reward_meter_mean": 0.9477019309997559, "reward_meter_std": 0.08215654641389847, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.948753833770752, "reward_repeat_soft_std": 0.035154614597558975, "reward_judge_quality_mean": 0.44624999165534973, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8052163124084473, "reward_total_composite_std": 0.03599754348397255} {"timestamp_utc": "2026-04-13T05:18:26Z", "mode": "train", "global_step": 2769, "epoch": 0.2781516825715721, "loss": 0.0232, "grad_norm": 12.392885208129883, "learning_rate": 1.6121212121212124e-06, "num_tokens": 5163168.0, "completions/mean_length": 25.625, "completions/min_length": 25.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.625, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.9702349901199341, "rewards/meter/std": 0.046526771038770676, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8926531076431274, "rewards/repeat_soft/std": 0.09270137548446655, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.8063710927963257, "rewards/total_composite/std": 0.021474801003932953, "reward": 0.8063710927963257, "reward_std": 0.021474795415997505, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07973424345254898, "sampling/sampling_logp_difference/max": 1.241170883178711, "sampling/importance_sampling_ratio/min": 0.2890455722808838, "sampling/importance_sampling_ratio/mean": 1.0235977172851562, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3570797797292471, "clip_ratio/low_mean": 0.024038462433964014, "clip_ratio/low_min": 0.024038462433964014, "clip_ratio/high_mean": 0.05887464340776205, "clip_ratio/high_max": 0.05887464340776205, "clip_ratio/region_mean": 0.08291310584172606, "reward_total_mean": 0.8063710927963257, "reward_meter_mean": 0.9702349901199341, "reward_meter_std": 0.046526771038770676, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8926531076431274, "reward_repeat_soft_std": 0.09270137548446655, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.8063710927963257, "reward_total_composite_std": 0.021474801003932953} {"timestamp_utc": "2026-04-13T05:18:34Z", "mode": "train", "global_step": 2770, "epoch": 0.2782521346057258, "loss": 0.0056, "grad_norm": 4.054328918457031, "learning_rate": 1.6090909090909092e-06, "num_tokens": 5165860.0, "completions/mean_length": 160.5, "completions/min_length": 148.0, "completions/max_length": 190.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 160.5, "completions/min_terminated_length": 148.0, "completions/max_terminated_length": 190.0, "rewards/meter/mean": 0.9914833307266235, "rewards/meter/std": 0.006693790666759014, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7335139513015747, "rewards/repeat_soft/std": 0.11838949471712112, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.771518886089325, "rewards/total_composite/std": 0.02897222526371479, "reward": 0.771518886089325, "reward_std": 0.02897222712635994, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05150426924228668, "sampling/sampling_logp_difference/max": 1.4639697074890137, "sampling/importance_sampling_ratio/min": 0.2313161939382553, "sampling/importance_sampling_ratio/mean": 1.003151297569275, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.270462304353714, "clip_ratio/low_mean": 0.01756340393330902, "clip_ratio/low_min": 0.01756340393330902, "clip_ratio/high_mean": 0.02528270846232772, "clip_ratio/high_max": 0.02528270846232772, "clip_ratio/region_mean": 0.04284611239563674, "reward_total_mean": 0.771518886089325, "reward_meter_mean": 0.9914833307266235, "reward_meter_std": 0.006693790666759014, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7335139513015747, "reward_repeat_soft_std": 0.11838949471712112, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.771518886089325, "reward_total_composite_std": 0.02897222526371479} {"timestamp_utc": "2026-04-13T05:18:40Z", "mode": "train", "global_step": 2771, "epoch": 0.27835258663987944, "loss": 0.0216, "grad_norm": 7.672449588775635, "learning_rate": 1.6060606060606063e-06, "num_tokens": 5167787.0, "completions/mean_length": 63.875, "completions/min_length": 59.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.875, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9875703454017639, "rewards/meter/std": 0.004340804647654295, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7434855103492737, "rewards/repeat_soft/std": 0.07668501138687134, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8003802299499512, "rewards/total_composite/std": 0.010103016160428524, "reward": 0.8003802299499512, "reward_std": 0.010103011503815651, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05867023393511772, "sampling/sampling_logp_difference/max": 1.4322788715362549, "sampling/importance_sampling_ratio/min": 0.2387641966342926, "sampling/importance_sampling_ratio/mean": 0.9981897473335266, "sampling/importance_sampling_ratio/max": 1.7789299488067627, "entropy": 0.2706174496561289, "clip_ratio/low_mean": 0.01579761365428567, "clip_ratio/low_min": 0.01579761365428567, "clip_ratio/high_mean": 0.03331391676329076, "clip_ratio/high_max": 0.03331391676329076, "clip_ratio/region_mean": 0.04911153041757643, "reward_total_mean": 0.8003802299499512, "reward_meter_mean": 0.9875703454017639, "reward_meter_std": 0.004340804647654295, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7434855103492737, "reward_repeat_soft_std": 0.07668501138687134, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8003802299499512, "reward_total_composite_std": 0.010103016160428524} {"timestamp_utc": "2026-04-13T05:18:48Z", "mode": "train", "global_step": 2772, "epoch": 0.27845303867403315, "loss": -0.0007, "grad_norm": 5.978781223297119, "learning_rate": 1.6030303030303033e-06, "num_tokens": 5170348.0, "completions/mean_length": 129.125, "completions/min_length": 111.0, "completions/max_length": 148.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 129.125, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 148.0, "rewards/meter/mean": 0.9910884499549866, "rewards/meter/std": 0.002121453871950507, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8041064739227295, "rewards/repeat_soft/std": 0.05346918851137161, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7832754850387573, "rewards/total_composite/std": 0.02644461952149868, "reward": 0.7832754850387573, "reward_std": 0.02644464001059532, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06109817326068878, "sampling/sampling_logp_difference/max": 2.151606559753418, "sampling/importance_sampling_ratio/min": 0.1162971705198288, "sampling/importance_sampling_ratio/mean": 1.0050499439239502, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.30244309082627296, "clip_ratio/low_mean": 0.01771742117125541, "clip_ratio/low_min": 0.01771742117125541, "clip_ratio/high_mean": 0.04372330056503415, "clip_ratio/high_max": 0.04372330056503415, "clip_ratio/region_mean": 0.06144072173628956, "reward_total_mean": 0.7832754850387573, "reward_meter_mean": 0.9910884499549866, "reward_meter_std": 0.002121453871950507, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8041064739227295, "reward_repeat_soft_std": 0.05346918851137161, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7832754850387573, "reward_total_composite_std": 0.02644461952149868} {"timestamp_utc": "2026-04-13T05:18:55Z", "mode": "train", "global_step": 2773, "epoch": 0.27855349070818686, "loss": -0.016, "grad_norm": 4.953784942626953, "learning_rate": 1.6000000000000001e-06, "num_tokens": 5172732.0, "completions/mean_length": 106.0, "completions/min_length": 101.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.0, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.9841978549957275, "rewards/meter/std": 0.010496334172785282, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9006292819976807, "rewards/repeat_soft/std": 0.014530448243021965, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.8277019262313843, "rewards/total_composite/std": 0.05458620563149452, "reward": 0.8277019262313843, "reward_std": 0.05458621308207512, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.060991670936346054, "sampling/sampling_logp_difference/max": 1.585597276687622, "sampling/importance_sampling_ratio/min": 0.20482541620731354, "sampling/importance_sampling_ratio/mean": 1.005916714668274, "sampling/importance_sampling_ratio/max": 1.8772820234298706, "entropy": 0.26838651672005653, "clip_ratio/low_mean": 0.04688334197271615, "clip_ratio/low_min": 0.04688334197271615, "clip_ratio/high_mean": 0.00657894741743803, "clip_ratio/high_max": 0.00657894741743803, "clip_ratio/region_mean": 0.05346228939015418, "reward_total_mean": 0.8277019262313843, "reward_meter_mean": 0.9841978549957275, "reward_meter_std": 0.010496334172785282, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9006292819976807, "reward_repeat_soft_std": 0.014530448243021965, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.8277019262313843, "reward_total_composite_std": 0.05458620563149452} {"timestamp_utc": "2026-04-13T05:19:01Z", "mode": "train", "global_step": 2774, "epoch": 0.2786539427423405, "loss": 0.0308, "grad_norm": 19.060644149780273, "learning_rate": 1.5969696969696972e-06, "num_tokens": 5174117.0, "completions/mean_length": 30.125, "completions/min_length": 26.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.125, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9733684062957764, "rewards/meter/std": 0.059076327830553055, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8170157670974731, "rewards/total_composite/std": 0.029365085065364838, "reward": 0.8170157670974731, "reward_std": 0.029365094378590584, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12186603248119354, "sampling/sampling_logp_difference/max": 1.1751365661621094, "sampling/importance_sampling_ratio/min": 0.3087767958641052, "sampling/importance_sampling_ratio/mean": 1.0101630687713623, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6467993259429932, "clip_ratio/low_mean": 0.008064515888690948, "clip_ratio/low_min": 0.008064515888690948, "clip_ratio/high_mean": 0.10984528623521328, "clip_ratio/high_max": 0.10984528623521328, "clip_ratio/region_mean": 0.11790980212390423, "reward_total_mean": 0.8170157670974731, "reward_meter_mean": 0.9733684062957764, "reward_meter_std": 0.059076327830553055, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8170157670974731, "reward_total_composite_std": 0.029365085065364838} {"timestamp_utc": "2026-04-13T05:19:08Z", "mode": "train", "global_step": 2775, "epoch": 0.2787543947764942, "loss": 0.0119, "grad_norm": 11.498627662658691, "learning_rate": 1.593939393939394e-06, "num_tokens": 5175838.0, "completions/mean_length": 43.125, "completions/min_length": 37.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.125, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9621858596801758, "rewards/meter/std": 0.008638124912977219, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9166879653930664, "rewards/repeat_soft/std": 0.11754637956619263, "rewards/judge_quality/mean": 0.39375001192092896, "rewards/judge_quality/std": 0.09941794723272324, "rewards/total_composite/mean": 0.792777419090271, "rewards/total_composite/std": 0.03811376914381981, "reward": 0.792777419090271, "reward_std": 0.03811377286911011, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0848725363612175, "sampling/sampling_logp_difference/max": 1.6333246231079102, "sampling/importance_sampling_ratio/min": 0.19527925550937653, "sampling/importance_sampling_ratio/mean": 0.9910237193107605, "sampling/importance_sampling_ratio/max": 1.8775757551193237, "entropy": 0.32332373782992363, "clip_ratio/low_mean": 0.014610390178859234, "clip_ratio/low_min": 0.014610390178859234, "clip_ratio/high_mean": 0.06185303768143058, "clip_ratio/high_max": 0.06185303768143058, "clip_ratio/region_mean": 0.07646342786028981, "reward_total_mean": 0.792777419090271, "reward_meter_mean": 0.9621858596801758, "reward_meter_std": 0.008638124912977219, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9166879653930664, "reward_repeat_soft_std": 0.11754637956619263, "reward_judge_quality_mean": 0.39375001192092896, "reward_judge_quality_std": 0.09941794723272324, "reward_total_composite_mean": 0.792777419090271, "reward_total_composite_std": 0.03811376914381981} {"timestamp_utc": "2026-04-13T05:19:19Z", "mode": "train", "global_step": 2776, "epoch": 0.2788548468106479, "loss": -0.2167, "grad_norm": 1.9237902164459229, "learning_rate": 1.590909090909091e-06, "num_tokens": 5178147.0, "completions/mean_length": 184.625, "completions/min_length": 132.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 137.85714721679688, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 147.0, "rewards/meter/mean": 0.9137817621231079, "rewards/meter/std": 0.11608556658029556, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9301693439483643, "rewards/repeat_soft/std": 0.04527668654918671, "rewards/judge_quality/mean": 0.3062499761581421, "rewards/judge_quality/std": 0.15999440848827362, "rewards/total_composite/mean": 0.6606307029724121, "rewards/total_composite/std": 0.2769988477230072, "reward": 0.6606307029724121, "reward_std": 0.2769988477230072, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08847904205322266, "sampling/sampling_logp_difference/max": 1.5713214874267578, "sampling/importance_sampling_ratio/min": 0.207770437002182, "sampling/importance_sampling_ratio/mean": 1.0104293823242188, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4413682483136654, "clip_ratio/low_mean": 0.014964789152145386, "clip_ratio/low_min": 0.014964789152145386, "clip_ratio/high_mean": 0.060965889133512974, "clip_ratio/high_max": 0.060965889133512974, "clip_ratio/region_mean": 0.07593067828565836, "reward_total_mean": 0.6606307029724121, "reward_meter_mean": 0.9137817621231079, "reward_meter_std": 0.11608556658029556, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.2651650309562683, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9301693439483643, "reward_repeat_soft_std": 0.04527668654918671, "reward_judge_quality_mean": 0.3062499761581421, "reward_judge_quality_std": 0.15999440848827362, "reward_total_composite_mean": 0.6606307029724121, "reward_total_composite_std": 0.2769988477230072} {"timestamp_utc": "2026-04-13T05:19:26Z", "mode": "train", "global_step": 2777, "epoch": 0.2789552988448016, "loss": 0.0744, "grad_norm": 11.100399017333984, "learning_rate": 1.5878787878787879e-06, "num_tokens": 5179769.0, "completions/mean_length": 49.75, "completions/min_length": 47.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.75, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9188846349716187, "rewards/meter/std": 0.16247789561748505, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9434641599655151, "rewards/repeat_soft/std": 0.03361007198691368, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.8025945425033569, "rewards/total_composite/std": 0.09660368412733078, "reward": 0.8025945425033569, "reward_std": 0.09660367667675018, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07863591611385345, "sampling/sampling_logp_difference/max": 1.7543857097625732, "sampling/importance_sampling_ratio/min": 0.17301349341869354, "sampling/importance_sampling_ratio/mean": 1.002152919769287, "sampling/importance_sampling_ratio/max": 1.9644505977630615, "entropy": 0.3200307320803404, "clip_ratio/low_mean": 0.008620689623057842, "clip_ratio/low_min": 0.008620689623057842, "clip_ratio/high_mean": 0.04908618680201471, "clip_ratio/high_max": 0.04908618680201471, "clip_ratio/region_mean": 0.05770687642507255, "reward_total_mean": 0.8025945425033569, "reward_meter_mean": 0.9188846349716187, "reward_meter_std": 0.16247789561748505, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9434641599655151, "reward_repeat_soft_std": 0.03361007198691368, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.8025945425033569, "reward_total_composite_std": 0.09660368412733078} {"timestamp_utc": "2026-04-13T05:19:33Z", "mode": "train", "global_step": 2778, "epoch": 0.2790557508789553, "loss": -0.0008, "grad_norm": 7.233282089233398, "learning_rate": 1.584848484848485e-06, "num_tokens": 5182135.0, "completions/mean_length": 112.75, "completions/min_length": 101.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.75, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.9505068063735962, "rewards/meter/std": 0.06773276627063751, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9617182016372681, "rewards/repeat_soft/std": 0.028193054720759392, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.7897748947143555, "rewards/total_composite/std": 0.03681601956486702, "reward": 0.7897748947143555, "reward_std": 0.03681601211428642, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10222121328115463, "sampling/sampling_logp_difference/max": 1.9331269264221191, "sampling/importance_sampling_ratio/min": 0.14469504356384277, "sampling/importance_sampling_ratio/mean": 1.003324031829834, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4929126687347889, "clip_ratio/low_mean": 0.04278419818729162, "clip_ratio/low_min": 0.04278419818729162, "clip_ratio/high_mean": 0.07422637660056353, "clip_ratio/high_max": 0.07422637660056353, "clip_ratio/region_mean": 0.11701057478785515, "reward_total_mean": 0.7897748947143555, "reward_meter_mean": 0.9505068063735962, "reward_meter_std": 0.06773276627063751, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9617182016372681, "reward_repeat_soft_std": 0.028193054720759392, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.7897748947143555, "reward_total_composite_std": 0.03681601956486702} {"timestamp_utc": "2026-04-13T05:19:39Z", "mode": "train", "global_step": 2779, "epoch": 0.279156202913109, "loss": -0.0415, "grad_norm": 16.15654182434082, "learning_rate": 1.5818181818181818e-06, "num_tokens": 5183521.0, "completions/mean_length": 26.25, "completions/min_length": 22.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.25, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.8067858219146729, "rewards/meter/std": 0.3468089997768402, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.946841299533844, "rewards/repeat_soft/std": 0.026096463203430176, "rewards/judge_quality/mean": 0.44999998807907104, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7427377700805664, "rewards/total_composite/std": 0.15511326491832733, "reward": 0.7427377700805664, "reward_std": 0.15511323511600494, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10358736664056778, "sampling/sampling_logp_difference/max": 1.4468345642089844, "sampling/importance_sampling_ratio/min": 0.2353139966726303, "sampling/importance_sampling_ratio/mean": 0.9881294965744019, "sampling/importance_sampling_ratio/max": 1.8307933807373047, "entropy": 0.37684784829616547, "clip_ratio/low_mean": 0.010489510837942362, "clip_ratio/low_min": 0.010489510837942362, "clip_ratio/high_mean": 0.06886574113741517, "clip_ratio/high_max": 0.06886574113741517, "clip_ratio/region_mean": 0.07935525197535753, "reward_total_mean": 0.7427377700805664, "reward_meter_mean": 0.8067858219146729, "reward_meter_std": 0.3468089997768402, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.946841299533844, "reward_repeat_soft_std": 0.026096463203430176, "reward_judge_quality_mean": 0.44999998807907104, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7427377700805664, "reward_total_composite_std": 0.15511326491832733} {"timestamp_utc": "2026-04-13T05:19:46Z", "mode": "train", "global_step": 2780, "epoch": 0.2792566549472627, "loss": 0.0266, "grad_norm": 12.70181941986084, "learning_rate": 1.5787878787878788e-06, "num_tokens": 5185284.0, "completions/mean_length": 55.375, "completions/min_length": 48.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.375, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9073053002357483, "rewards/meter/std": 0.17429088056087494, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.971589982509613, "rewards/repeat_soft/std": 0.02769479714334011, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7814463973045349, "rewards/total_composite/std": 0.07720412313938141, "reward": 0.7814463973045349, "reward_std": 0.07720411568880081, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09312061965465546, "sampling/sampling_logp_difference/max": 1.2766733169555664, "sampling/importance_sampling_ratio/min": 0.27896377444267273, "sampling/importance_sampling_ratio/mean": 1.0031189918518066, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44876939058303833, "clip_ratio/low_mean": 0.012198464944958687, "clip_ratio/low_min": 0.012198464944958687, "clip_ratio/high_mean": 0.06394164869561791, "clip_ratio/high_max": 0.06394164869561791, "clip_ratio/region_mean": 0.0761401136405766, "reward_total_mean": 0.7814463973045349, "reward_meter_mean": 0.9073053002357483, "reward_meter_std": 0.17429088056087494, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.971589982509613, "reward_repeat_soft_std": 0.02769479714334011, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7814463973045349, "reward_total_composite_std": 0.07720412313938141} {"timestamp_utc": "2026-04-13T05:19:52Z", "mode": "train", "global_step": 2781, "epoch": 0.27935710698141636, "loss": 0.017, "grad_norm": 6.646609306335449, "learning_rate": 1.5757575757575759e-06, "num_tokens": 5187158.0, "completions/mean_length": 68.25, "completions/min_length": 64.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.25, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.987444281578064, "rewards/meter/std": 0.006400790065526962, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8958626985549927, "rewards/repeat_soft/std": 0.05180830508470535, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.8496862649917603, "rewards/total_composite/std": 0.06757467240095139, "reward": 0.8496862649917603, "reward_std": 0.06757467240095139, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08466918766498566, "sampling/sampling_logp_difference/max": 2.0367045402526855, "sampling/importance_sampling_ratio/min": 0.13045792281627655, "sampling/importance_sampling_ratio/mean": 0.9934457540512085, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.35294803604483604, "clip_ratio/low_mean": 0.054806909523904324, "clip_ratio/low_min": 0.054806909523904324, "clip_ratio/high_mean": 0.018568442203104496, "clip_ratio/high_max": 0.018568442203104496, "clip_ratio/region_mean": 0.07337535172700882, "reward_total_mean": 0.8496862649917603, "reward_meter_mean": 0.987444281578064, "reward_meter_std": 0.006400790065526962, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8958626985549927, "reward_repeat_soft_std": 0.05180830508470535, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.8496862649917603, "reward_total_composite_std": 0.06757467240095139} {"timestamp_utc": "2026-04-13T05:19:58Z", "mode": "train", "global_step": 2782, "epoch": 0.27945755901557007, "loss": -0.0023, "grad_norm": 7.8282952308654785, "learning_rate": 1.572727272727273e-06, "num_tokens": 5188608.0, "completions/mean_length": 26.25, "completions/min_length": 25.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.25, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.9885401725769043, "rewards/meter/std": 0.00493130786344409, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8238431215286255, "rewards/total_composite/std": 0.003346055978909135, "reward": 0.8238431215286255, "reward_std": 0.003346055746078491, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03680044040083885, "sampling/sampling_logp_difference/max": 0.9727926254272461, "sampling/importance_sampling_ratio/min": 0.3780258893966675, "sampling/importance_sampling_ratio/mean": 1.0008219480514526, "sampling/importance_sampling_ratio/max": 1.4742597341537476, "entropy": 0.23366787284612656, "clip_ratio/low_mean": 0.004629629664123058, "clip_ratio/low_min": 0.004629629664123058, "clip_ratio/high_mean": 0.0192592591047287, "clip_ratio/high_max": 0.0192592591047287, "clip_ratio/region_mean": 0.023888888768851757, "reward_total_mean": 0.8238431215286255, "reward_meter_mean": 0.9885401725769043, "reward_meter_std": 0.00493130786344409, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8238431215286255, "reward_total_composite_std": 0.003346055978909135} {"timestamp_utc": "2026-04-13T05:20:10Z", "mode": "train", "global_step": 2783, "epoch": 0.2795580110497238, "loss": -0.2013, "grad_norm": 2.4218454360961914, "learning_rate": 1.56969696969697e-06, "num_tokens": 5191100.0, "completions/mean_length": 182.5, "completions/min_length": 111.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 135.42857360839844, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 169.0, "rewards/meter/mean": 0.6784658432006836, "rewards/meter/std": 0.19945786893367767, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.12817397713661194, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9002213478088379, "rewards/repeat_soft/std": 0.044833917170763016, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.23092593252658844, "rewards/total_composite/mean": 0.6033955812454224, "rewards/total_composite/std": 0.2582830488681793, "reward": 0.6033955812454224, "reward_std": 0.25828301906585693, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0890151709318161, "sampling/sampling_logp_difference/max": 1.8105617761611938, "sampling/importance_sampling_ratio/min": 0.16356222331523895, "sampling/importance_sampling_ratio/mean": 1.01333487033844, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.30626892298460007, "clip_ratio/low_mean": 0.006656804587692022, "clip_ratio/low_min": 0.006656804587692022, "clip_ratio/high_mean": 0.05963137838989496, "clip_ratio/high_max": 0.05963137838989496, "clip_ratio/region_mean": 0.06628818297758698, "reward_total_mean": 0.6033955812454224, "reward_meter_mean": 0.6784658432006836, "reward_meter_std": 0.19945786893367767, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.12817397713661194, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9002213478088379, "reward_repeat_soft_std": 0.044833917170763016, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.23092593252658844, "reward_total_composite_mean": 0.6033955812454224, "reward_total_composite_std": 0.2582830488681793} {"timestamp_utc": "2026-04-13T05:20:21Z", "mode": "train", "global_step": 2784, "epoch": 0.27965846308387743, "loss": -0.2229, "grad_norm": 1.4265632629394531, "learning_rate": 1.566666666666667e-06, "num_tokens": 5193539.0, "completions/mean_length": 181.875, "completions/min_length": 128.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 134.71429443359375, "completions/min_terminated_length": 128.0, "completions/max_terminated_length": 141.0, "rewards/meter/mean": 0.9893880486488342, "rewards/meter/std": 0.002779321512207389, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.6880583763122559, "rewards/repeat_soft/std": 0.07865574955940247, "rewards/judge_quality/mean": 0.4737499952316284, "rewards/judge_quality/std": 0.2546110451221466, "rewards/total_composite/mean": 0.7198830246925354, "rewards/total_composite/std": 0.2962455749511719, "reward": 0.7198830246925354, "reward_std": 0.29624560475349426, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06898322701454163, "sampling/sampling_logp_difference/max": 2.3540897369384766, "sampling/importance_sampling_ratio/min": 0.0949799194931984, "sampling/importance_sampling_ratio/mean": 0.9926388263702393, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2256650235503912, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.053851418197155, "clip_ratio/high_max": 0.053851418197155, "clip_ratio/region_mean": 0.053851418197155, "reward_total_mean": 0.7198830246925354, "reward_meter_mean": 0.9893880486488342, "reward_meter_std": 0.002779321512207389, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.6880583763122559, "reward_repeat_soft_std": 0.07865574955940247, "reward_judge_quality_mean": 0.4737499952316284, "reward_judge_quality_std": 0.2546110451221466, "reward_total_composite_mean": 0.7198830246925354, "reward_total_composite_std": 0.2962455749511719} {"timestamp_utc": "2026-04-13T05:20:27Z", "mode": "train", "global_step": 2785, "epoch": 0.27975891511803114, "loss": -0.0182, "grad_norm": 8.10684585571289, "learning_rate": 1.5636363636363638e-06, "num_tokens": 5195088.0, "completions/mean_length": 32.625, "completions/min_length": 31.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.625, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9738737940788269, "rewards/meter/std": 0.035678680986166, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9613315463066101, "rewards/repeat_soft/std": 0.05079552158713341, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8126263618469238, "rewards/total_composite/std": 0.013192742131650448, "reward": 0.8126263618469238, "reward_std": 0.013192734681069851, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08855567872524261, "sampling/sampling_logp_difference/max": 1.2752857208251953, "sampling/importance_sampling_ratio/min": 0.279351145029068, "sampling/importance_sampling_ratio/mean": 0.990599513053894, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.41767672449350357, "clip_ratio/low_mean": 0.0587732158601284, "clip_ratio/low_min": 0.0587732158601284, "clip_ratio/high_mean": 0.04569128900766373, "clip_ratio/high_max": 0.04569128900766373, "clip_ratio/region_mean": 0.10446450486779213, "reward_total_mean": 0.8126263618469238, "reward_meter_mean": 0.9738737940788269, "reward_meter_std": 0.035678680986166, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9613315463066101, "reward_repeat_soft_std": 0.05079552158713341, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8126263618469238, "reward_total_composite_std": 0.013192742131650448} {"timestamp_utc": "2026-04-13T05:20:33Z", "mode": "train", "global_step": 2786, "epoch": 0.27985936715218485, "loss": 0.0144, "grad_norm": 11.414382934570312, "learning_rate": 1.5606060606060609e-06, "num_tokens": 5196652.0, "completions/mean_length": 30.5, "completions/min_length": 29.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.5, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.8730583190917969, "rewards/meter/std": 0.12849825620651245, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9833248853683472, "rewards/repeat_soft/std": 0.022231420502066612, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.19078317284584045, "rewards/total_composite/mean": 0.7983337640762329, "rewards/total_composite/std": 0.05234956368803978, "reward": 0.7983337640762329, "reward_std": 0.05234955996274948, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09393543004989624, "sampling/sampling_logp_difference/max": 1.705682635307312, "sampling/importance_sampling_ratio/min": 0.1816483438014984, "sampling/importance_sampling_ratio/mean": 0.9929487705230713, "sampling/importance_sampling_ratio/max": 1.603054165840149, "entropy": 0.3860113974660635, "clip_ratio/low_mean": 0.042935825418680906, "clip_ratio/low_min": 0.042935825418680906, "clip_ratio/high_mean": 0.02515294821932912, "clip_ratio/high_max": 0.02515294821932912, "clip_ratio/region_mean": 0.06808877363801003, "reward_total_mean": 0.7983337640762329, "reward_meter_mean": 0.8730583190917969, "reward_meter_std": 0.12849825620651245, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9833248853683472, "reward_repeat_soft_std": 0.022231420502066612, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.19078317284584045, "reward_total_composite_mean": 0.7983337640762329, "reward_total_composite_std": 0.05234956368803978} {"timestamp_utc": "2026-04-13T05:20:40Z", "mode": "train", "global_step": 2787, "epoch": 0.2799598191863385, "loss": 0.0572, "grad_norm": 6.834659099578857, "learning_rate": 1.5575757575757577e-06, "num_tokens": 5199036.0, "completions/mean_length": 115.0, "completions/min_length": 102.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.0, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.9045104384422302, "rewards/meter/std": 0.21105283498764038, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9693593978881836, "rewards/repeat_soft/std": 0.013046098873019218, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7799656391143799, "rewards/total_composite/std": 0.0957799181342125, "reward": 0.7799656391143799, "reward_std": 0.09577992558479309, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0986330434679985, "sampling/sampling_logp_difference/max": 2.3697714805603027, "sampling/importance_sampling_ratio/min": 0.09350208938121796, "sampling/importance_sampling_ratio/mean": 1.007130742073059, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.449093222618103, "clip_ratio/low_mean": 0.009920635260641575, "clip_ratio/low_min": 0.009920635260641575, "clip_ratio/high_mean": 0.07733542658388615, "clip_ratio/high_max": 0.07733542658388615, "clip_ratio/region_mean": 0.08725606184452772, "reward_total_mean": 0.7799656391143799, "reward_meter_mean": 0.9045104384422302, "reward_meter_std": 0.21105283498764038, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9693593978881836, "reward_repeat_soft_std": 0.013046098873019218, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7799656391143799, "reward_total_composite_std": 0.0957799181342125} {"timestamp_utc": "2026-04-13T05:20:52Z", "mode": "train", "global_step": 2788, "epoch": 0.2800602712204922, "loss": -0.2249, "grad_norm": 1.2561486959457397, "learning_rate": 1.5545454545454547e-06, "num_tokens": 5201464.0, "completions/mean_length": 186.5, "completions/min_length": 137.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 140.0, "completions/min_terminated_length": 137.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.9633241891860962, "rewards/meter/std": 0.06637665629386902, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7697137594223022, "rewards/repeat_soft/std": 0.08238503336906433, "rewards/judge_quality/mean": 0.33124998211860657, "rewards/judge_quality/std": 0.13715866208076477, "rewards/total_composite/mean": 0.6833928823471069, "rewards/total_composite/std": 0.276738703250885, "reward": 0.6833928823471069, "reward_std": 0.2767386734485626, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07828013598918915, "sampling/sampling_logp_difference/max": 2.9791250228881836, "sampling/importance_sampling_ratio/min": 0.05083729326725006, "sampling/importance_sampling_ratio/mean": 1.0061326026916504, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.29223491437733173, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06338589126244187, "clip_ratio/high_max": 0.06338589126244187, "clip_ratio/region_mean": 0.06338589126244187, "reward_total_mean": 0.6833928823471069, "reward_meter_mean": 0.9633241891860962, "reward_meter_std": 0.06637665629386902, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7697137594223022, "reward_repeat_soft_std": 0.08238503336906433, "reward_judge_quality_mean": 0.33124998211860657, "reward_judge_quality_std": 0.13715866208076477, "reward_total_composite_mean": 0.6833928823471069, "reward_total_composite_std": 0.276738703250885} {"timestamp_utc": "2026-04-13T05:20:59Z", "mode": "train", "global_step": 2789, "epoch": 0.2801607232546459, "loss": -0.0227, "grad_norm": 4.505472183227539, "learning_rate": 1.5515151515151516e-06, "num_tokens": 5203917.0, "completions/mean_length": 121.625, "completions/min_length": 101.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 121.625, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.9315522313117981, "rewards/meter/std": 0.04794292896986008, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6228787899017334, "rewards/repeat_soft/std": 0.06043000891804695, "rewards/judge_quality/mean": 0.2799999713897705, "rewards/judge_quality/std": 0.09304375946521759, "rewards/total_composite/mean": 0.6967363357543945, "rewards/total_composite/std": 0.038539737462997437, "reward": 0.6967363357543945, "reward_std": 0.038539737462997437, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0446215495467186, "sampling/sampling_logp_difference/max": 1.4397532939910889, "sampling/importance_sampling_ratio/min": 0.2369862198829651, "sampling/importance_sampling_ratio/mean": 1.00284743309021, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.19662146642804146, "clip_ratio/low_mean": 0.016230184119194746, "clip_ratio/low_min": 0.016230184119194746, "clip_ratio/high_mean": 0.028029863722622395, "clip_ratio/high_max": 0.028029863722622395, "clip_ratio/region_mean": 0.04426004784181714, "reward_total_mean": 0.6967363357543945, "reward_meter_mean": 0.9315522313117981, "reward_meter_std": 0.04794292896986008, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6228787899017334, "reward_repeat_soft_std": 0.06043000891804695, "reward_judge_quality_mean": 0.2799999713897705, "reward_judge_quality_std": 0.09304375946521759, "reward_total_composite_mean": 0.6967363357543945, "reward_total_composite_std": 0.038539737462997437} {"timestamp_utc": "2026-04-13T05:21:06Z", "mode": "train", "global_step": 2790, "epoch": 0.2802611752887996, "loss": 0.0365, "grad_norm": 7.264495372772217, "learning_rate": 1.5484848484848486e-06, "num_tokens": 5205682.0, "completions/mean_length": 67.625, "completions/min_length": 60.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.625, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9849272966384888, "rewards/meter/std": 0.0017431833548471332, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8622279763221741, "rewards/repeat_soft/std": 0.11826497316360474, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.1940544992685318, "rewards/total_composite/mean": 0.8189401030540466, "rewards/total_composite/std": 0.06149974837899208, "reward": 0.8189401030540466, "reward_std": 0.06149975210428238, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09644034504890442, "sampling/sampling_logp_difference/max": 2.4046528339385986, "sampling/importance_sampling_ratio/min": 0.09029684215784073, "sampling/importance_sampling_ratio/mean": 1.0041319131851196, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43992427736520767, "clip_ratio/low_mean": 0.05254841921851039, "clip_ratio/low_min": 0.05254841921851039, "clip_ratio/high_mean": 0.02742537297308445, "clip_ratio/high_max": 0.02742537297308445, "clip_ratio/region_mean": 0.07997379219159484, "reward_total_mean": 0.8189401030540466, "reward_meter_mean": 0.9849272966384888, "reward_meter_std": 0.0017431833548471332, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8622279763221741, "reward_repeat_soft_std": 0.11826497316360474, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.1940544992685318, "reward_total_composite_mean": 0.8189401030540466, "reward_total_composite_std": 0.06149974837899208} {"timestamp_utc": "2026-04-13T05:21:12Z", "mode": "train", "global_step": 2791, "epoch": 0.2803616273229533, "loss": 0.0068, "grad_norm": 7.408412456512451, "learning_rate": 1.5454545454545454e-06, "num_tokens": 5207442.0, "completions/mean_length": 58.0, "completions/min_length": 53.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9710894227027893, "rewards/meter/std": 0.04446195811033249, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9391031265258789, "rewards/repeat_soft/std": 0.03452586010098457, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.808025598526001, "rewards/total_composite/std": 0.022463154047727585, "reward": 0.808025598526001, "reward_std": 0.022463154047727585, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08782435953617096, "sampling/sampling_logp_difference/max": 1.2895565032958984, "sampling/importance_sampling_ratio/min": 0.27539288997650146, "sampling/importance_sampling_ratio/mean": 1.000291347503662, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4212075546383858, "clip_ratio/low_mean": 0.021130953449755907, "clip_ratio/low_min": 0.021130953449755907, "clip_ratio/high_mean": 0.06199410231783986, "clip_ratio/high_max": 0.06199410231783986, "clip_ratio/region_mean": 0.08312505576759577, "reward_total_mean": 0.808025598526001, "reward_meter_mean": 0.9710894227027893, "reward_meter_std": 0.04446195811033249, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9391031265258789, "reward_repeat_soft_std": 0.03452586010098457, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.808025598526001, "reward_total_composite_std": 0.022463154047727585} {"timestamp_utc": "2026-04-13T05:21:19Z", "mode": "train", "global_step": 2792, "epoch": 0.280462079357107, "loss": 0.0173, "grad_norm": 12.626140594482422, "learning_rate": 1.5424242424242425e-06, "num_tokens": 5209250.0, "completions/mean_length": 73.0, "completions/min_length": 64.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9538689255714417, "rewards/meter/std": 0.04966919124126434, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9348142743110657, "rewards/repeat_soft/std": 0.03118608519434929, "rewards/judge_quality/mean": 0.48499998450279236, "rewards/judge_quality/std": 0.16017846763134003, "rewards/total_composite/mean": 0.8182224035263062, "rewards/total_composite/std": 0.05453001707792282, "reward": 0.8182224035263062, "reward_std": 0.054530028253793716, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1001409962773323, "sampling/sampling_logp_difference/max": 1.2585031986236572, "sampling/importance_sampling_ratio/min": 0.308285117149353, "sampling/importance_sampling_ratio/mean": 0.9976012110710144, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43073683977127075, "clip_ratio/low_mean": 0.064399050665088, "clip_ratio/low_min": 0.064399050665088, "clip_ratio/high_mean": 0.025686338543891907, "clip_ratio/high_max": 0.025686338543891907, "clip_ratio/region_mean": 0.0900853892089799, "reward_total_mean": 0.8182224035263062, "reward_meter_mean": 0.9538689255714417, "reward_meter_std": 0.04966919124126434, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9348142743110657, "reward_repeat_soft_std": 0.03118608519434929, "reward_judge_quality_mean": 0.48499998450279236, "reward_judge_quality_std": 0.16017846763134003, "reward_total_composite_mean": 0.8182224035263062, "reward_total_composite_std": 0.05453001707792282} {"timestamp_utc": "2026-04-13T05:21:30Z", "mode": "train", "global_step": 2793, "epoch": 0.2805625313912607, "loss": -0.2527, "grad_norm": 1.819507122039795, "learning_rate": 1.5393939393939395e-06, "num_tokens": 5211681.0, "completions/mean_length": 246.875, "completions/min_length": 152.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 158.5, "completions/min_terminated_length": 152.0, "completions/max_terminated_length": 167.0, "rewards/meter/mean": 0.776089608669281, "rewards/meter/std": 0.2708463668823242, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.3500283360481262, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9055358171463013, "rewards/repeat_soft/std": 0.0614376924932003, "rewards/judge_quality/mean": 0.32749998569488525, "rewards/judge_quality/std": 0.17127670347690582, "rewards/total_composite/mean": 0.5825937986373901, "rewards/total_composite/std": 0.361113041639328, "reward": 0.5825937986373901, "reward_std": 0.3611130118370056, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09777595847845078, "sampling/sampling_logp_difference/max": 2.9641127586364746, "sampling/importance_sampling_ratio/min": 0.05160623788833618, "sampling/importance_sampling_ratio/mean": 1.0191004276275635, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2723800279200077, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.05846612434834242, "clip_ratio/high_max": 0.05846612434834242, "clip_ratio/region_mean": 0.05846612434834242, "reward_total_mean": 0.5825937986373901, "reward_meter_mean": 0.776089608669281, "reward_meter_std": 0.2708463668823242, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.3500283360481262, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9055358171463013, "reward_repeat_soft_std": 0.0614376924932003, "reward_judge_quality_mean": 0.32749998569488525, "reward_judge_quality_std": 0.17127670347690582, "reward_total_composite_mean": 0.5825937986373901, "reward_total_composite_std": 0.361113041639328} {"timestamp_utc": "2026-04-13T05:21:42Z", "mode": "train", "global_step": 2794, "epoch": 0.28066298342541435, "loss": -0.2219, "grad_norm": 1.5146719217300415, "learning_rate": 1.5363636363636364e-06, "num_tokens": 5213972.0, "completions/mean_length": 186.375, "completions/min_length": 130.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 139.85714721679688, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 148.0, "rewards/meter/mean": 0.9045276641845703, "rewards/meter/std": 0.2507310211658478, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8044068217277527, "rewards/repeat_soft/std": 0.08012524247169495, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.7368531227111816, "rewards/total_composite/std": 0.14522208273410797, "reward": 0.7368531227111816, "reward_std": 0.14522208273410797, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06735877692699432, "sampling/sampling_logp_difference/max": 1.9671320915222168, "sampling/importance_sampling_ratio/min": 0.13985736668109894, "sampling/importance_sampling_ratio/mean": 1.0037778615951538, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.26049042120575905, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06265277042984962, "clip_ratio/high_max": 0.06265277042984962, "clip_ratio/region_mean": 0.06265277042984962, "reward_total_mean": 0.7368531227111816, "reward_meter_mean": 0.9045276641845703, "reward_meter_std": 0.2507310211658478, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8044068217277527, "reward_repeat_soft_std": 0.08012524247169495, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.7368531227111816, "reward_total_composite_std": 0.14522208273410797} {"timestamp_utc": "2026-04-13T05:21:48Z", "mode": "train", "global_step": 2795, "epoch": 0.28076343545956806, "loss": 0.0419, "grad_norm": 8.104190826416016, "learning_rate": 1.5333333333333334e-06, "num_tokens": 5215783.0, "completions/mean_length": 62.375, "completions/min_length": 58.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.375, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9660104513168335, "rewards/meter/std": 0.007332310080528259, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8706874847412109, "rewards/repeat_soft/std": 0.09607361257076263, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.798898458480835, "rewards/total_composite/std": 0.009659326635301113, "reward": 0.798898458480835, "reward_std": 0.00965932197868824, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06540461629629135, "sampling/sampling_logp_difference/max": 1.3516541719436646, "sampling/importance_sampling_ratio/min": 0.2588117718696594, "sampling/importance_sampling_ratio/mean": 0.9963778853416443, "sampling/importance_sampling_ratio/max": 1.8802032470703125, "entropy": 0.23744477704167366, "clip_ratio/low_mean": 0.015584186650812626, "clip_ratio/low_min": 0.015584186650812626, "clip_ratio/high_mean": 0.03204879583790898, "clip_ratio/high_max": 0.03204879583790898, "clip_ratio/region_mean": 0.04763298248872161, "reward_total_mean": 0.798898458480835, "reward_meter_mean": 0.9660104513168335, "reward_meter_std": 0.007332310080528259, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8706874847412109, "reward_repeat_soft_std": 0.09607361257076263, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.798898458480835, "reward_total_composite_std": 0.009659326635301113} {"timestamp_utc": "2026-04-13T05:21:55Z", "mode": "train", "global_step": 2796, "epoch": 0.28086388749372176, "loss": -0.035, "grad_norm": 17.957889556884766, "learning_rate": 1.5303030303030302e-06, "num_tokens": 5217304.0, "completions/mean_length": 39.125, "completions/min_length": 34.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.125, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.8637416958808899, "rewards/meter/std": 0.2925844192504883, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9114012718200684, "rewards/repeat_soft/std": 0.06409254670143127, "rewards/judge_quality/mean": 0.5137500166893005, "rewards/judge_quality/std": 0.21387162804603577, "rewards/total_composite/mean": 0.7839488983154297, "rewards/total_composite/std": 0.0916459858417511, "reward": 0.7839488983154297, "reward_std": 0.09164600074291229, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10725463926792145, "sampling/sampling_logp_difference/max": 1.7625818252563477, "sampling/importance_sampling_ratio/min": 0.17160123586654663, "sampling/importance_sampling_ratio/mean": 0.9872173070907593, "sampling/importance_sampling_ratio/max": 1.7256040573120117, "entropy": 0.48750946298241615, "clip_ratio/low_mean": 0.026333789341151714, "clip_ratio/low_min": 0.026333789341151714, "clip_ratio/high_mean": 0.07019849703647196, "clip_ratio/high_max": 0.07019849703647196, "clip_ratio/region_mean": 0.09653228637762368, "reward_total_mean": 0.7839488983154297, "reward_meter_mean": 0.8637416958808899, "reward_meter_std": 0.2925844192504883, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9114012718200684, "reward_repeat_soft_std": 0.06409254670143127, "reward_judge_quality_mean": 0.5137500166893005, "reward_judge_quality_std": 0.21387162804603577, "reward_total_composite_mean": 0.7839488983154297, "reward_total_composite_std": 0.0916459858417511} {"timestamp_utc": "2026-04-13T05:22:01Z", "mode": "train", "global_step": 2797, "epoch": 0.2809643395278754, "loss": 0.0335, "grad_norm": 12.669023513793945, "learning_rate": 1.5272727272727275e-06, "num_tokens": 5219026.0, "completions/mean_length": 55.25, "completions/min_length": 50.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.25, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9871362447738647, "rewards/meter/std": 0.0026896665804088116, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9623511433601379, "rewards/repeat_soft/std": 0.028101298958063126, "rewards/judge_quality/mean": 0.4762499928474426, "rewards/judge_quality/std": 0.19167961180210114, "rewards/total_composite/mean": 0.7300045490264893, "rewards/total_composite/std": 0.30072253942489624, "reward": 0.7300045490264893, "reward_std": 0.30072250962257385, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08018383383750916, "sampling/sampling_logp_difference/max": 1.2198028564453125, "sampling/importance_sampling_ratio/min": 0.2952883839607239, "sampling/importance_sampling_ratio/mean": 1.0010912418365479, "sampling/importance_sampling_ratio/max": 1.7593116760253906, "entropy": 0.42993640527129173, "clip_ratio/low_mean": 0.010964912362396717, "clip_ratio/low_min": 0.010964912362396717, "clip_ratio/high_mean": 0.07926032692193985, "clip_ratio/high_max": 0.07926032692193985, "clip_ratio/region_mean": 0.09022523928433657, "reward_total_mean": 0.7300045490264893, "reward_meter_mean": 0.9871362447738647, "reward_meter_std": 0.0026896665804088116, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9623511433601379, "reward_repeat_soft_std": 0.028101298958063126, "reward_judge_quality_mean": 0.4762499928474426, "reward_judge_quality_std": 0.19167961180210114, "reward_total_composite_mean": 0.7300045490264893, "reward_total_composite_std": 0.30072253942489624} {"timestamp_utc": "2026-04-13T05:22:07Z", "mode": "train", "global_step": 2798, "epoch": 0.2810647915620291, "loss": 0.0021, "grad_norm": 8.130976676940918, "learning_rate": 1.5242424242424245e-06, "num_tokens": 5220328.0, "completions/mean_length": 22.75, "completions/min_length": 20.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.75, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.9623428583145142, "rewards/meter/std": 0.041059281677007675, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9303127527236938, "rewards/repeat_soft/std": 0.027077898383140564, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8032105565071106, "rewards/total_composite/std": 0.01974773406982422, "reward": 0.8032105565071106, "reward_std": 0.019747747108340263, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08018811792135239, "sampling/sampling_logp_difference/max": 1.1413660049438477, "sampling/importance_sampling_ratio/min": 0.3193824291229248, "sampling/importance_sampling_ratio/mean": 1.0075665712356567, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3909139037132263, "clip_ratio/low_mean": 0.03456959826871753, "clip_ratio/low_min": 0.03456959826871753, "clip_ratio/high_mean": 0.03818140272051096, "clip_ratio/high_max": 0.03818140272051096, "clip_ratio/region_mean": 0.07275100098922849, "reward_total_mean": 0.8032105565071106, "reward_meter_mean": 0.9623428583145142, "reward_meter_std": 0.041059281677007675, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9303127527236938, "reward_repeat_soft_std": 0.027077898383140564, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8032105565071106, "reward_total_composite_std": 0.01974773406982422} {"timestamp_utc": "2026-04-13T05:22:14Z", "mode": "train", "global_step": 2799, "epoch": 0.28116524359618283, "loss": -0.0028, "grad_norm": 9.628860473632812, "learning_rate": 1.5212121212121214e-06, "num_tokens": 5222159.0, "completions/mean_length": 75.875, "completions/min_length": 67.0, "completions/max_length": 89.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.875, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 89.0, "rewards/meter/mean": 0.8912566900253296, "rewards/meter/std": 0.19766885042190552, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9405277371406555, "rewards/repeat_soft/std": 0.046995654702186584, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.13695022463798523, "rewards/total_composite/mean": 0.7947432994842529, "rewards/total_composite/std": 0.10847941786050797, "reward": 0.7947432994842529, "reward_std": 0.10847943276166916, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11027192324399948, "sampling/sampling_logp_difference/max": 1.6323111057281494, "sampling/importance_sampling_ratio/min": 0.19547727704048157, "sampling/importance_sampling_ratio/mean": 1.0148124694824219, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45146995037794113, "clip_ratio/low_mean": 0.028716216795146465, "clip_ratio/low_min": 0.028716216795146465, "clip_ratio/high_mean": 0.07674364279955626, "clip_ratio/high_max": 0.07674364279955626, "clip_ratio/region_mean": 0.10545985959470272, "reward_total_mean": 0.7947432994842529, "reward_meter_mean": 0.8912566900253296, "reward_meter_std": 0.19766885042190552, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9405277371406555, "reward_repeat_soft_std": 0.046995654702186584, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.13695022463798523, "reward_total_composite_mean": 0.7947432994842529, "reward_total_composite_std": 0.10847941786050797} {"timestamp_utc": "2026-04-13T05:22:20Z", "mode": "train", "global_step": 2800, "epoch": 0.2812656956303365, "loss": 0.0272, "grad_norm": 7.487646579742432, "learning_rate": 1.5181818181818184e-06, "num_tokens": 5224019.0, "completions/mean_length": 57.5, "completions/min_length": 53.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.5, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9622859954833984, "rewards/meter/std": 0.0424666628241539, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9653510451316833, "rewards/repeat_soft/std": 0.031102988868951797, "rewards/judge_quality/mean": 0.35874998569488525, "rewards/judge_quality/std": 0.12205823510885239, "rewards/total_composite/mean": 0.7871887683868408, "rewards/total_composite/std": 0.032127149403095245, "reward": 0.7871887683868408, "reward_std": 0.03212713822722435, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08870996534824371, "sampling/sampling_logp_difference/max": 1.5105361938476562, "sampling/importance_sampling_ratio/min": 0.22079156339168549, "sampling/importance_sampling_ratio/mean": 1.0074912309646606, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.46614134311676025, "clip_ratio/low_mean": 0.04481152771040797, "clip_ratio/low_min": 0.04481152771040797, "clip_ratio/high_mean": 0.04361540684476495, "clip_ratio/high_max": 0.04361540684476495, "clip_ratio/region_mean": 0.08842693455517292, "reward_total_mean": 0.7871887683868408, "reward_meter_mean": 0.9622859954833984, "reward_meter_std": 0.0424666628241539, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9653510451316833, "reward_repeat_soft_std": 0.031102988868951797, "reward_judge_quality_mean": 0.35874998569488525, "reward_judge_quality_std": 0.12205823510885239, "reward_total_composite_mean": 0.7871887683868408, "reward_total_composite_std": 0.032127149403095245} {"timestamp_utc": "2026-04-13T05:23:11Z", "mode": "eval", "global_step": 2800, "epoch": 0.2812656956303365, "eval_loss": NaN, "eval_runtime": 50.2236, "eval_samples_per_second": 1.593, "eval_steps_per_second": 0.199, "eval_num_tokens": 5224019.0, "eval_completions/mean_length": 99.5125, "eval_completions/min_length": 43.6, "eval_completions/max_length": 197.4, "eval_completions/clipped_ratio": 0.0125, "eval_completions/mean_terminated_length": 94.66250076293946, "eval_completions/min_terminated_length": 43.6, "eval_completions/max_terminated_length": 164.0, "eval_rewards/meter/mean": 0.9178199589252471, "eval_rewards/meter/std": 0.14824384041130542, "eval_rewards/count_adherence/mean": 0.9883333265781402, "eval_rewards/count_adherence/std": 0.028114382922649384, "eval_rewards/hard_gate/mean": 1.0, "eval_rewards/hard_gate/std": 0.0, "eval_rewards/repeat_soft/mean": 0.903091162443161, "eval_rewards/repeat_soft/std": 0.0907859917730093, "eval_rewards/judge_quality/mean": 0.41374999582767485, "eval_rewards/judge_quality/std": 0.11736709931865334, "eval_rewards/total_composite/mean": 0.7757030844688415, "eval_rewards/total_composite/std": 0.08892021737992764, "eval_reward": 0.7757030844688415, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.03957518264651298, "eval_sampling/sampling_logp_difference/max": 0.8191468715667725, "eval_sampling/importance_sampling_ratio/min": 0.44661614000797273, "eval_sampling/importance_sampling_ratio/mean": 1.0055998325347901, "eval_sampling/importance_sampling_ratio/max": 1.3294743537902831, "eval_entropy": 0.37042124271392823, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7757030844688415, "eval_reward_meter_mean": 0.9178199589252471, "eval_reward_meter_std": 0.14824384041130542, "eval_reward_count_adherence_mean": 0.9883333265781402, "eval_reward_count_adherence_std": 0.028114382922649384, "eval_reward_hard_gate_mean": 1.0, "eval_reward_hard_gate_std": 0.0, "eval_reward_repeat_soft_mean": 0.903091162443161, "eval_reward_repeat_soft_std": 0.0907859917730093, "eval_reward_judge_quality_mean": 0.41374999582767485, "eval_reward_judge_quality_std": 0.11736709931865334, "eval_reward_total_composite_mean": 0.7757030844688415, "eval_reward_total_composite_std": 0.08892021737992764} {"timestamp_utc": "2026-04-13T05:23:21Z", "mode": "train", "global_step": 2801, "epoch": 0.2813661476644902, "loss": 0.0258, "grad_norm": 5.520373821258545, "learning_rate": 1.5151515151515152e-06, "num_tokens": 5226385.0, "completions/mean_length": 126.75, "completions/min_length": 119.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.75, "completions/min_terminated_length": 119.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.9898878335952759, "rewards/meter/std": 0.004849364515393972, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6902818083763123, "rewards/repeat_soft/std": 0.11424562335014343, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7904777526855469, "rewards/total_composite/std": 0.009794776327908039, "reward": 0.7904777526855469, "reward_std": 0.009794753976166248, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06150023639202118, "sampling/sampling_logp_difference/max": 3.0396728515625, "sampling/importance_sampling_ratio/min": 0.04785054177045822, "sampling/importance_sampling_ratio/mean": 0.9974093437194824, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.26039119996130466, "clip_ratio/low_mean": 0.03190571768209338, "clip_ratio/low_min": 0.03190571768209338, "clip_ratio/high_mean": 0.020734001416713, "clip_ratio/high_max": 0.020734001416713, "clip_ratio/region_mean": 0.05263971909880638, "reward_total_mean": 0.7904777526855469, "reward_meter_mean": 0.9898878335952759, "reward_meter_std": 0.004849364515393972, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6902818083763123, "reward_repeat_soft_std": 0.11424562335014343, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7904777526855469, "reward_total_composite_std": 0.009794776327908039} {"timestamp_utc": "2026-04-13T05:23:28Z", "mode": "train", "global_step": 2802, "epoch": 0.2814665996986439, "loss": -0.0011, "grad_norm": 7.235088348388672, "learning_rate": 1.5121212121212123e-06, "num_tokens": 5228500.0, "completions/mean_length": 95.375, "completions/min_length": 86.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.375, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.9945151805877686, "rewards/meter/std": 0.0019450357649475336, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9500610828399658, "rewards/repeat_soft/std": 0.02947319485247135, "rewards/judge_quality/mean": 0.6825000047683716, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.8972879648208618, "rewards/total_composite/std": 0.07194343209266663, "reward": 0.8972879648208618, "reward_std": 0.07194343209266663, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08610619604587555, "sampling/sampling_logp_difference/max": 1.328468680381775, "sampling/importance_sampling_ratio/min": 0.2648825943470001, "sampling/importance_sampling_ratio/mean": 1.0112370252609253, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4022531397640705, "clip_ratio/low_mean": 0.033853634260594845, "clip_ratio/low_min": 0.033853634260594845, "clip_ratio/high_mean": 0.04899974539875984, "clip_ratio/high_max": 0.04899974539875984, "clip_ratio/region_mean": 0.08285337965935469, "reward_total_mean": 0.8972879648208618, "reward_meter_mean": 0.9945151805877686, "reward_meter_std": 0.0019450357649475336, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9500610828399658, "reward_repeat_soft_std": 0.02947319485247135, "reward_judge_quality_mean": 0.6825000047683716, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.8972879648208618, "reward_total_composite_std": 0.07194343209266663} {"timestamp_utc": "2026-04-13T05:23:36Z", "mode": "train", "global_step": 2803, "epoch": 0.2815670517327976, "loss": 0.0171, "grad_norm": 5.848292827606201, "learning_rate": 1.5090909090909091e-06, "num_tokens": 5230971.0, "completions/mean_length": 136.875, "completions/min_length": 130.0, "completions/max_length": 155.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.875, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 155.0, "rewards/meter/mean": 0.9555919170379639, "rewards/meter/std": 0.052980631589889526, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7561564445495605, "rewards/repeat_soft/std": 0.10265693813562393, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7785069942474365, "rewards/total_composite/std": 0.02945019118487835, "reward": 0.7785069942474365, "reward_std": 0.029450180009007454, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05160069465637207, "sampling/sampling_logp_difference/max": 1.8895728588104248, "sampling/importance_sampling_ratio/min": 0.1511363536119461, "sampling/importance_sampling_ratio/mean": 0.9989540576934814, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.20658898167312145, "clip_ratio/low_mean": 0.013817455619573593, "clip_ratio/low_min": 0.013817455619573593, "clip_ratio/high_mean": 0.022824802377726883, "clip_ratio/high_max": 0.022824802377726883, "clip_ratio/region_mean": 0.036642257997300476, "reward_total_mean": 0.7785069942474365, "reward_meter_mean": 0.9555919170379639, "reward_meter_std": 0.052980631589889526, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7561564445495605, "reward_repeat_soft_std": 0.10265693813562393, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7785069942474365, "reward_total_composite_std": 0.02945019118487835} {"timestamp_utc": "2026-04-13T05:23:44Z", "mode": "train", "global_step": 2804, "epoch": 0.28166750376695127, "loss": 0.0374, "grad_norm": 4.754316806793213, "learning_rate": 1.5060606060606062e-06, "num_tokens": 5233367.0, "completions/mean_length": 138.5, "completions/min_length": 115.0, "completions/max_length": 151.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 138.5, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 151.0, "rewards/meter/mean": 0.9913433790206909, "rewards/meter/std": 0.003312828252092004, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8169997930526733, "rewards/repeat_soft/std": 0.06702694296836853, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.7873045206069946, "rewards/total_composite/std": 0.03266964107751846, "reward": 0.7873045206069946, "reward_std": 0.032669637352228165, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06979656964540482, "sampling/sampling_logp_difference/max": 1.767979621887207, "sampling/importance_sampling_ratio/min": 0.17067746818065643, "sampling/importance_sampling_ratio/mean": 0.9986568093299866, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3076497595757246, "clip_ratio/low_mean": 0.012329932302236557, "clip_ratio/low_min": 0.012329932302236557, "clip_ratio/high_mean": 0.055539064574986696, "clip_ratio/high_max": 0.055539064574986696, "clip_ratio/region_mean": 0.06786899687722325, "reward_total_mean": 0.7873045206069946, "reward_meter_mean": 0.9913433790206909, "reward_meter_std": 0.003312828252092004, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8169997930526733, "reward_repeat_soft_std": 0.06702694296836853, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.7873045206069946, "reward_total_composite_std": 0.03266964107751846} {"timestamp_utc": "2026-04-13T05:23:50Z", "mode": "train", "global_step": 2805, "epoch": 0.281767955801105, "loss": -0.0448, "grad_norm": 17.7427921295166, "learning_rate": 1.5030303030303032e-06, "num_tokens": 5234857.0, "completions/mean_length": 28.25, "completions/min_length": 23.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.25, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9467277526855469, "rewards/meter/std": 0.07588482648134232, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9553571343421936, "rewards/repeat_soft/std": 0.013339361175894737, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.8020632266998291, "rewards/total_composite/std": 0.037169329822063446, "reward": 0.8020632266998291, "reward_std": 0.03716934099793434, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10093103349208832, "sampling/sampling_logp_difference/max": 1.708374261856079, "sampling/importance_sampling_ratio/min": 0.1811600625514984, "sampling/importance_sampling_ratio/mean": 1.0159157514572144, "sampling/importance_sampling_ratio/max": 1.8252328634262085, "entropy": 0.5212702415883541, "clip_ratio/low_mean": 0.023800600320100784, "clip_ratio/low_min": 0.023800600320100784, "clip_ratio/high_mean": 0.07726254779845476, "clip_ratio/high_max": 0.07726254779845476, "clip_ratio/region_mean": 0.10106314811855555, "reward_total_mean": 0.8020632266998291, "reward_meter_mean": 0.9467277526855469, "reward_meter_std": 0.07588482648134232, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9553571343421936, "reward_repeat_soft_std": 0.013339361175894737, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.8020632266998291, "reward_total_composite_std": 0.037169329822063446} {"timestamp_utc": "2026-04-13T05:23:57Z", "mode": "train", "global_step": 2806, "epoch": 0.2818684078352587, "loss": 0.0336, "grad_norm": 6.251988887786865, "learning_rate": 1.5e-06, "num_tokens": 5237111.0, "completions/mean_length": 117.75, "completions/min_length": 113.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.75, "completions/min_terminated_length": 113.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.9887968897819519, "rewards/meter/std": 0.005822585429996252, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8133807182312012, "rewards/repeat_soft/std": 0.07457120716571808, "rewards/judge_quality/mean": 0.5149999856948853, "rewards/judge_quality/std": 0.1810288280248642, "rewards/total_composite/mean": 0.8307967185974121, "rewards/total_composite/std": 0.057503584772348404, "reward": 0.8307967185974121, "reward_std": 0.057503581047058105, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06737960129976273, "sampling/sampling_logp_difference/max": 1.6531322002410889, "sampling/importance_sampling_ratio/min": 0.19144931435585022, "sampling/importance_sampling_ratio/mean": 1.0040298700332642, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3349607288837433, "clip_ratio/low_mean": 0.04799178009852767, "clip_ratio/low_min": 0.04799178009852767, "clip_ratio/high_mean": 0.020863793790340424, "clip_ratio/high_max": 0.020863793790340424, "clip_ratio/region_mean": 0.0688555738888681, "reward_total_mean": 0.8307967185974121, "reward_meter_mean": 0.9887968897819519, "reward_meter_std": 0.005822585429996252, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8133807182312012, "reward_repeat_soft_std": 0.07457120716571808, "reward_judge_quality_mean": 0.5149999856948853, "reward_judge_quality_std": 0.1810288280248642, "reward_total_composite_mean": 0.8307967185974121, "reward_total_composite_std": 0.057503584772348404} {"timestamp_utc": "2026-04-13T05:24:03Z", "mode": "train", "global_step": 2807, "epoch": 0.28196885986941234, "loss": 0.0393, "grad_norm": 13.819191932678223, "learning_rate": 1.496969696969697e-06, "num_tokens": 5238513.0, "completions/mean_length": 26.25, "completions/min_length": 24.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.25, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9839701056480408, "rewards/meter/std": 0.005958087742328644, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9135501384735107, "rewards/repeat_soft/std": 0.05095406249165535, "rewards/judge_quality/mean": 0.4087499976158142, "rewards/judge_quality/std": 0.10507649928331375, "rewards/total_composite/mean": 0.8067665100097656, "rewards/total_composite/std": 0.03455930948257446, "reward": 0.8067665100097656, "reward_std": 0.03455932065844536, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08887214958667755, "sampling/sampling_logp_difference/max": 1.4239535331726074, "sampling/importance_sampling_ratio/min": 0.24076028168201447, "sampling/importance_sampling_ratio/mean": 1.0034481287002563, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4235237315297127, "clip_ratio/low_mean": 0.004310344811528921, "clip_ratio/low_min": 0.004310344811528921, "clip_ratio/high_mean": 0.07408827636390924, "clip_ratio/high_max": 0.07408827636390924, "clip_ratio/region_mean": 0.07839862117543817, "reward_total_mean": 0.8067665100097656, "reward_meter_mean": 0.9839701056480408, "reward_meter_std": 0.005958087742328644, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9135501384735107, "reward_repeat_soft_std": 0.05095406249165535, "reward_judge_quality_mean": 0.4087499976158142, "reward_judge_quality_std": 0.10507649928331375, "reward_total_composite_mean": 0.8067665100097656, "reward_total_composite_std": 0.03455930948257446} {"timestamp_utc": "2026-04-13T05:24:09Z", "mode": "train", "global_step": 2808, "epoch": 0.28206931190356604, "loss": 0.0499, "grad_norm": 10.719515800476074, "learning_rate": 1.493939393939394e-06, "num_tokens": 5240203.0, "completions/mean_length": 37.25, "completions/min_length": 33.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.25, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9392072558403015, "rewards/meter/std": 0.11725357174873352, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9788916110992432, "rewards/repeat_soft/std": 0.025404034182429314, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.8715324401855469, "rewards/total_composite/std": 0.07549955695867538, "reward": 0.8715324401855469, "reward_std": 0.07549955695867538, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08583419770002365, "sampling/sampling_logp_difference/max": 1.530332326889038, "sampling/importance_sampling_ratio/min": 0.22340022027492523, "sampling/importance_sampling_ratio/mean": 1.0116437673568726, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2865962013602257, "clip_ratio/low_mean": 0.059071412310004234, "clip_ratio/low_min": 0.059071412310004234, "clip_ratio/high_mean": 0.031860902439802885, "clip_ratio/high_max": 0.031860902439802885, "clip_ratio/region_mean": 0.09093231474980712, "reward_total_mean": 0.8715324401855469, "reward_meter_mean": 0.9392072558403015, "reward_meter_std": 0.11725357174873352, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9788916110992432, "reward_repeat_soft_std": 0.025404034182429314, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.8715324401855469, "reward_total_composite_std": 0.07549955695867538} {"timestamp_utc": "2026-04-13T05:24:16Z", "mode": "train", "global_step": 2809, "epoch": 0.28216976393771975, "loss": 0.0304, "grad_norm": 6.403763294219971, "learning_rate": 1.490909090909091e-06, "num_tokens": 5242484.0, "completions/mean_length": 118.125, "completions/min_length": 113.0, "completions/max_length": 127.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.125, "completions/min_terminated_length": 113.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.9516207575798035, "rewards/meter/std": 0.056318361312150955, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.894853949546814, "rewards/repeat_soft/std": 0.060904163867235184, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7937147617340088, "rewards/total_composite/std": 0.023600200191140175, "reward": 0.7937147617340088, "reward_std": 0.02360019087791443, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08004482835531235, "sampling/sampling_logp_difference/max": 1.826888084411621, "sampling/importance_sampling_ratio/min": 0.16091354191303253, "sampling/importance_sampling_ratio/mean": 1.008601427078247, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.32096628472208977, "clip_ratio/low_mean": 0.02052444312721491, "clip_ratio/low_min": 0.02052444312721491, "clip_ratio/high_mean": 0.03431468550115824, "clip_ratio/high_max": 0.03431468550115824, "clip_ratio/region_mean": 0.054839128628373146, "reward_total_mean": 0.7937147617340088, "reward_meter_mean": 0.9516207575798035, "reward_meter_std": 0.056318361312150955, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.894853949546814, "reward_repeat_soft_std": 0.060904163867235184, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7937147617340088, "reward_total_composite_std": 0.023600200191140175} {"timestamp_utc": "2026-04-13T05:24:23Z", "mode": "train", "global_step": 2810, "epoch": 0.2822702159718734, "loss": 0.018, "grad_norm": 7.397765159606934, "learning_rate": 1.4878787878787878e-06, "num_tokens": 5244224.0, "completions/mean_length": 57.5, "completions/min_length": 54.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.5, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9813587069511414, "rewards/meter/std": 0.007913145236670971, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9093925952911377, "rewards/repeat_soft/std": 0.06257116049528122, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.8648006916046143, "rewards/total_composite/std": 0.07707158476114273, "reward": 0.8648006916046143, "reward_std": 0.07707156985998154, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06384525448083878, "sampling/sampling_logp_difference/max": 1.1256120204925537, "sampling/importance_sampling_ratio/min": 0.32445383071899414, "sampling/importance_sampling_ratio/mean": 0.9966738224029541, "sampling/importance_sampling_ratio/max": 1.8369613885879517, "entropy": 0.3516788948327303, "clip_ratio/low_mean": 0.045637457398697734, "clip_ratio/low_min": 0.045637457398697734, "clip_ratio/high_mean": 0.013202388305217028, "clip_ratio/high_max": 0.013202388305217028, "clip_ratio/region_mean": 0.05883984570391476, "reward_total_mean": 0.8648006916046143, "reward_meter_mean": 0.9813587069511414, "reward_meter_std": 0.007913145236670971, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9093925952911377, "reward_repeat_soft_std": 0.06257116049528122, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.8648006916046143, "reward_total_composite_std": 0.07707158476114273} {"timestamp_utc": "2026-04-13T05:24:29Z", "mode": "train", "global_step": 2811, "epoch": 0.2823706680060271, "loss": 0.0051, "grad_norm": 8.546649932861328, "learning_rate": 1.484848484848485e-06, "num_tokens": 5245998.0, "completions/mean_length": 69.75, "completions/min_length": 68.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.75, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9464114904403687, "rewards/meter/std": 0.06058825924992561, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8397506475448608, "rewards/repeat_soft/std": 0.09915482252836227, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.1249857097864151, "rewards/total_composite/mean": 0.7656102180480957, "rewards/total_composite/std": 0.04477936029434204, "reward": 0.7656102180480957, "reward_std": 0.04477935656905174, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06765458732843399, "sampling/sampling_logp_difference/max": 2.3515992164611816, "sampling/importance_sampling_ratio/min": 0.095216765999794, "sampling/importance_sampling_ratio/mean": 1.0004528760910034, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2585088275372982, "clip_ratio/low_mean": 0.020140665350481868, "clip_ratio/low_min": 0.020140665350481868, "clip_ratio/high_mean": 0.037278654519468546, "clip_ratio/high_max": 0.037278654519468546, "clip_ratio/region_mean": 0.057419319869950414, "reward_total_mean": 0.7656102180480957, "reward_meter_mean": 0.9464114904403687, "reward_meter_std": 0.06058825924992561, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8397506475448608, "reward_repeat_soft_std": 0.09915482252836227, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.1249857097864151, "reward_total_composite_mean": 0.7656102180480957, "reward_total_composite_std": 0.04477936029434204} {"timestamp_utc": "2026-04-13T05:24:35Z", "mode": "train", "global_step": 2812, "epoch": 0.2824711200401808, "loss": 0.0207, "grad_norm": 12.813514709472656, "learning_rate": 1.481818181818182e-06, "num_tokens": 5247585.0, "completions/mean_length": 29.375, "completions/min_length": 27.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.375, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.9858216047286987, "rewards/meter/std": 0.010560998693108559, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9438823461532593, "rewards/repeat_soft/std": 0.024053387343883514, "rewards/judge_quality/mean": 0.6187499761581421, "rewards/judge_quality/std": 0.24976776540279388, "rewards/total_composite/mean": 0.8736329674720764, "rewards/total_composite/std": 0.076995849609375, "reward": 0.8736329674720764, "reward_std": 0.0769958421587944, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12416081875562668, "sampling/sampling_logp_difference/max": 2.0740156173706055, "sampling/importance_sampling_ratio/min": 0.1256800889968872, "sampling/importance_sampling_ratio/mean": 1.0210530757904053, "sampling/importance_sampling_ratio/max": 1.921737790107727, "entropy": 0.5066388286650181, "clip_ratio/low_mean": 0.0925863403826952, "clip_ratio/low_min": 0.0925863403826952, "clip_ratio/high_mean": 0.0417861407622695, "clip_ratio/high_max": 0.0417861407622695, "clip_ratio/region_mean": 0.1343724811449647, "reward_total_mean": 0.8736329674720764, "reward_meter_mean": 0.9858216047286987, "reward_meter_std": 0.010560998693108559, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9438823461532593, "reward_repeat_soft_std": 0.024053387343883514, "reward_judge_quality_mean": 0.6187499761581421, "reward_judge_quality_std": 0.24976776540279388, "reward_total_composite_mean": 0.8736329674720764, "reward_total_composite_std": 0.076995849609375} {"timestamp_utc": "2026-04-13T05:24:47Z", "mode": "train", "global_step": 2813, "epoch": 0.28257157207433453, "loss": -0.1385, "grad_norm": 2.444577932357788, "learning_rate": 1.478787878787879e-06, "num_tokens": 5249234.0, "completions/mean_length": 116.125, "completions/min_length": 49.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 59.57143020629883, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.7565187215805054, "rewards/meter/std": 0.39694148302078247, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9464108347892761, "rewards/repeat_soft/std": 0.04145852103829384, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.13452960550785065, "rewards/total_composite/mean": 0.6663244962692261, "rewards/total_composite/std": 0.2918635308742523, "reward": 0.6663244962692261, "reward_std": 0.2918635308742523, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10565643012523651, "sampling/sampling_logp_difference/max": 1.4218659400939941, "sampling/importance_sampling_ratio/min": 0.24126343429088593, "sampling/importance_sampling_ratio/mean": 1.0070983171463013, "sampling/importance_sampling_ratio/max": 1.8497390747070312, "entropy": 0.558190606534481, "clip_ratio/low_mean": 0.013671875, "clip_ratio/low_min": 0.013671875, "clip_ratio/high_mean": 0.10458880756050348, "clip_ratio/high_max": 0.10458880756050348, "clip_ratio/region_mean": 0.11826068256050348, "reward_total_mean": 0.6663244962692261, "reward_meter_mean": 0.7565187215805054, "reward_meter_std": 0.39694148302078247, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9464108347892761, "reward_repeat_soft_std": 0.04145852103829384, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.13452960550785065, "reward_total_composite_mean": 0.6663244962692261, "reward_total_composite_std": 0.2918635308742523} {"timestamp_utc": "2026-04-13T05:24:53Z", "mode": "train", "global_step": 2814, "epoch": 0.2826720241084882, "loss": 0.0444, "grad_norm": 17.82896614074707, "learning_rate": 1.475757575757576e-06, "num_tokens": 5250655.0, "completions/mean_length": 22.625, "completions/min_length": 21.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.625, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.9882485866546631, "rewards/meter/std": 0.004283030051738024, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.773751974105835, "rewards/repeat_soft/std": 0.1424817442893982, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.8025870323181152, "rewards/total_composite/std": 0.013815384358167648, "reward": 0.8025870323181152, "reward_std": 0.013815375044941902, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06936896592378616, "sampling/sampling_logp_difference/max": 0.9870920181274414, "sampling/importance_sampling_ratio/min": 0.3726588189601898, "sampling/importance_sampling_ratio/mean": 1.0155198574066162, "sampling/importance_sampling_ratio/max": 1.6123309135437012, "entropy": 0.3825165815651417, "clip_ratio/low_mean": 0.04255849262699485, "clip_ratio/low_min": 0.04255849262699485, "clip_ratio/high_mean": 0.017316017765551805, "clip_ratio/high_max": 0.017316017765551805, "clip_ratio/region_mean": 0.059874510392546654, "reward_total_mean": 0.8025870323181152, "reward_meter_mean": 0.9882485866546631, "reward_meter_std": 0.004283030051738024, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.773751974105835, "reward_repeat_soft_std": 0.1424817442893982, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.8025870323181152, "reward_total_composite_std": 0.013815384358167648} {"timestamp_utc": "2026-04-13T05:24:59Z", "mode": "train", "global_step": 2815, "epoch": 0.2827724761426419, "loss": 0.024, "grad_norm": 7.081674575805664, "learning_rate": 1.4727272727272728e-06, "num_tokens": 5252555.0, "completions/mean_length": 59.5, "completions/min_length": 57.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.5, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.7994940280914307, "rewards/meter/std": 0.26366671919822693, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8858224153518677, "rewards/repeat_soft/std": 0.05184835195541382, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.7431045174598694, "rewards/total_composite/std": 0.14309701323509216, "reward": 0.7431045174598694, "reward_std": 0.14309701323509216, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.058788422495126724, "sampling/sampling_logp_difference/max": 1.5235390663146973, "sampling/importance_sampling_ratio/min": 0.21793922781944275, "sampling/importance_sampling_ratio/mean": 1.0064786672592163, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2521089222282171, "clip_ratio/low_mean": 0.023213292937725782, "clip_ratio/low_min": 0.023213292937725782, "clip_ratio/high_mean": 0.03984728408977389, "clip_ratio/high_max": 0.03984728408977389, "clip_ratio/region_mean": 0.06306057702749968, "reward_total_mean": 0.7431045174598694, "reward_meter_mean": 0.7994940280914307, "reward_meter_std": 0.26366671919822693, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8858224153518677, "reward_repeat_soft_std": 0.05184835195541382, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.7431045174598694, "reward_total_composite_std": 0.14309701323509216} {"timestamp_utc": "2026-04-13T05:25:06Z", "mode": "train", "global_step": 2816, "epoch": 0.2828729281767956, "loss": 0.0212, "grad_norm": 6.511547088623047, "learning_rate": 1.4696969696969698e-06, "num_tokens": 5254325.0, "completions/mean_length": 46.25, "completions/min_length": 43.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.25, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.963716983795166, "rewards/meter/std": 0.01860002428293228, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9007259607315063, "rewards/repeat_soft/std": 0.08372373878955841, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.0843462198972702, "rewards/total_composite/mean": 0.7892452478408813, "rewards/total_composite/std": 0.028781678527593613, "reward": 0.7892452478408813, "reward_std": 0.028781671077013016, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05719510465860367, "sampling/sampling_logp_difference/max": 1.1103248596191406, "sampling/importance_sampling_ratio/min": 0.32945194840431213, "sampling/importance_sampling_ratio/mean": 1.0122034549713135, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2986285574734211, "clip_ratio/low_mean": 0.016493055736646056, "clip_ratio/low_min": 0.016493055736646056, "clip_ratio/high_mean": 0.04859154252335429, "clip_ratio/high_max": 0.04859154252335429, "clip_ratio/region_mean": 0.06508459826000035, "reward_total_mean": 0.7892452478408813, "reward_meter_mean": 0.963716983795166, "reward_meter_std": 0.01860002428293228, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9007259607315063, "reward_repeat_soft_std": 0.08372373878955841, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.0843462198972702, "reward_total_composite_mean": 0.7892452478408813, "reward_total_composite_std": 0.028781678527593613} {"timestamp_utc": "2026-04-13T05:25:14Z", "mode": "train", "global_step": 2817, "epoch": 0.28297338021094925, "loss": 0.0637, "grad_norm": 4.662861347198486, "learning_rate": 1.4666666666666669e-06, "num_tokens": 5257507.0, "completions/mean_length": 201.75, "completions/min_length": 188.0, "completions/max_length": 237.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 201.75, "completions/min_terminated_length": 188.0, "completions/max_terminated_length": 237.0, "rewards/meter/mean": 0.9895047545433044, "rewards/meter/std": 0.003494562581181526, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7618316411972046, "rewards/repeat_soft/std": 0.0697464868426323, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7943353056907654, "rewards/total_composite/std": 0.012277464382350445, "reward": 0.7943353056907654, "reward_std": 0.012277476489543915, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05776376277208328, "sampling/sampling_logp_difference/max": 1.7756781578063965, "sampling/importance_sampling_ratio/min": 0.16936855018138885, "sampling/importance_sampling_ratio/mean": 0.9997823238372803, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23588185384869576, "clip_ratio/low_mean": 0.008632845478132367, "clip_ratio/low_min": 0.008632845478132367, "clip_ratio/high_mean": 0.04190096631646156, "clip_ratio/high_max": 0.04190096631646156, "clip_ratio/region_mean": 0.05053381179459393, "reward_total_mean": 0.7943353056907654, "reward_meter_mean": 0.9895047545433044, "reward_meter_std": 0.003494562581181526, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7618316411972046, "reward_repeat_soft_std": 0.0697464868426323, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7943353056907654, "reward_total_composite_std": 0.012277464382350445} {"timestamp_utc": "2026-04-13T05:25:21Z", "mode": "train", "global_step": 2818, "epoch": 0.28307383224510296, "loss": 0.0273, "grad_norm": 10.187671661376953, "learning_rate": 1.4636363636363637e-06, "num_tokens": 5259148.0, "completions/mean_length": 45.125, "completions/min_length": 39.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.125, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.9404186010360718, "rewards/meter/std": 0.052013594657182693, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8916339874267578, "rewards/repeat_soft/std": 0.12310509383678436, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7692267894744873, "rewards/total_composite/std": 0.033017054200172424, "reward": 0.7692267894744873, "reward_std": 0.033017054200172424, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07766872644424438, "sampling/sampling_logp_difference/max": 1.1293158531188965, "sampling/importance_sampling_ratio/min": 0.3232543468475342, "sampling/importance_sampling_ratio/mean": 1.0074585676193237, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3483987729996443, "clip_ratio/low_mean": 0.03707060334272683, "clip_ratio/low_min": 0.03707060334272683, "clip_ratio/high_mean": 0.03355662291869521, "clip_ratio/high_max": 0.03355662291869521, "clip_ratio/region_mean": 0.07062722626142204, "reward_total_mean": 0.7692267894744873, "reward_meter_mean": 0.9404186010360718, "reward_meter_std": 0.052013594657182693, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8916339874267578, "reward_repeat_soft_std": 0.12310509383678436, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7692267894744873, "reward_total_composite_std": 0.033017054200172424} {"timestamp_utc": "2026-04-13T05:25:28Z", "mode": "train", "global_step": 2819, "epoch": 0.28317428427925667, "loss": -0.0027, "grad_norm": 7.789121627807617, "learning_rate": 1.4606060606060608e-06, "num_tokens": 5261114.0, "completions/mean_length": 83.75, "completions/min_length": 64.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.75, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.5669337511062622, "rewards/meter/std": 0.4161481559276581, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.945412814617157, "rewards/repeat_soft/std": 0.03700482100248337, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.1249857097864151, "rewards/total_composite/mean": 0.5929114818572998, "rewards/total_composite/std": 0.18479260802268982, "reward": 0.5929114818572998, "reward_std": 0.18479260802268982, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08107313513755798, "sampling/sampling_logp_difference/max": 1.8404450416564941, "sampling/importance_sampling_ratio/min": 0.15874676406383514, "sampling/importance_sampling_ratio/mean": 1.0099191665649414, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4114357829093933, "clip_ratio/low_mean": 0.046938205137848854, "clip_ratio/low_min": 0.046938205137848854, "clip_ratio/high_mean": 0.040017335675656796, "clip_ratio/high_max": 0.040017335675656796, "clip_ratio/region_mean": 0.08695554081350565, "reward_total_mean": 0.5929114818572998, "reward_meter_mean": 0.5669337511062622, "reward_meter_std": 0.4161481559276581, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.945412814617157, "reward_repeat_soft_std": 0.03700482100248337, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.1249857097864151, "reward_total_composite_mean": 0.5929114818572998, "reward_total_composite_std": 0.18479260802268982} {"timestamp_utc": "2026-04-13T05:25:35Z", "mode": "train", "global_step": 2820, "epoch": 0.2832747363134103, "loss": 0.0476, "grad_norm": 12.524832725524902, "learning_rate": 1.4575757575757576e-06, "num_tokens": 5262541.0, "completions/mean_length": 23.375, "completions/min_length": 19.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.375, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.9795520305633545, "rewards/meter/std": 0.010883050970733166, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.817548394203186, "rewards/total_composite/std": 0.004263665992766619, "reward": 0.817548394203186, "reward_std": 0.004263669718056917, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07276802510023117, "sampling/sampling_logp_difference/max": 0.9812992811203003, "sampling/importance_sampling_ratio/min": 0.37482380867004395, "sampling/importance_sampling_ratio/mean": 1.0016099214553833, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3370245061814785, "clip_ratio/low_mean": 0.03704578196629882, "clip_ratio/low_min": 0.03704578196629882, "clip_ratio/high_mean": 0.02595238061621785, "clip_ratio/high_max": 0.02595238061621785, "clip_ratio/region_mean": 0.06299816258251667, "reward_total_mean": 0.817548394203186, "reward_meter_mean": 0.9795520305633545, "reward_meter_std": 0.010883050970733166, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.817548394203186, "reward_total_composite_std": 0.004263665992766619} {"timestamp_utc": "2026-04-13T05:25:42Z", "mode": "train", "global_step": 2821, "epoch": 0.28337518834756403, "loss": 0.0357, "grad_norm": 10.941219329833984, "learning_rate": 1.4545454545454546e-06, "num_tokens": 5264239.0, "completions/mean_length": 55.25, "completions/min_length": 48.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.25, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.987128496170044, "rewards/meter/std": 0.0021792438346892595, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9898334741592407, "rewards/repeat_soft/std": 0.012965784408152103, "rewards/judge_quality/mean": 0.8575000166893005, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.9504411816596985, "rewards/total_composite/std": 0.05301584675908089, "reward": 0.9504411816596985, "reward_std": 0.05301584303379059, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07660629600286484, "sampling/sampling_logp_difference/max": 1.5261766910552979, "sampling/importance_sampling_ratio/min": 0.21736514568328857, "sampling/importance_sampling_ratio/mean": 0.9954212307929993, "sampling/importance_sampling_ratio/max": 1.708534836769104, "entropy": 0.3027539439499378, "clip_ratio/low_mean": 0.004310344811528921, "clip_ratio/low_min": 0.004310344811528921, "clip_ratio/high_mean": 0.0519154230132699, "clip_ratio/high_max": 0.0519154230132699, "clip_ratio/region_mean": 0.05622576782479882, "reward_total_mean": 0.9504411816596985, "reward_meter_mean": 0.987128496170044, "reward_meter_std": 0.0021792438346892595, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9898334741592407, "reward_repeat_soft_std": 0.012965784408152103, "reward_judge_quality_mean": 0.8575000166893005, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.9504411816596985, "reward_total_composite_std": 0.05301584675908089} {"timestamp_utc": "2026-04-13T05:25:48Z", "mode": "train", "global_step": 2822, "epoch": 0.28347564038171774, "loss": 0.0742, "grad_norm": 10.269784927368164, "learning_rate": 1.4515151515151515e-06, "num_tokens": 5265950.0, "completions/mean_length": 52.875, "completions/min_length": 49.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.875, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.8432890176773071, "rewards/meter/std": 0.29368263483047485, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9797099828720093, "rewards/repeat_soft/std": 0.020788518711924553, "rewards/judge_quality/mean": 0.42124998569488525, "rewards/judge_quality/std": 0.06998724490404129, "rewards/total_composite/mean": 0.7538260221481323, "rewards/total_composite/std": 0.12888994812965393, "reward": 0.7538260221481323, "reward_std": 0.12888994812965393, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07992403954267502, "sampling/sampling_logp_difference/max": 1.6606178283691406, "sampling/importance_sampling_ratio/min": 0.1900215446949005, "sampling/importance_sampling_ratio/mean": 0.9958721399307251, "sampling/importance_sampling_ratio/max": 1.9294378757476807, "entropy": 0.38123833388090134, "clip_ratio/low_mean": 0.008474576054140925, "clip_ratio/low_min": 0.008474576054140925, "clip_ratio/high_mean": 0.06846495205536485, "clip_ratio/high_max": 0.06846495205536485, "clip_ratio/region_mean": 0.07693952810950577, "reward_total_mean": 0.7538260221481323, "reward_meter_mean": 0.8432890176773071, "reward_meter_std": 0.29368263483047485, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9797099828720093, "reward_repeat_soft_std": 0.020788518711924553, "reward_judge_quality_mean": 0.42124998569488525, "reward_judge_quality_std": 0.06998724490404129, "reward_total_composite_mean": 0.7538260221481323, "reward_total_composite_std": 0.12888994812965393} {"timestamp_utc": "2026-04-13T05:26:00Z", "mode": "train", "global_step": 2823, "epoch": 0.2835760924158714, "loss": -0.2121, "grad_norm": 2.0184895992279053, "learning_rate": 1.4484848484848485e-06, "num_tokens": 5268212.0, "completions/mean_length": 171.75, "completions/min_length": 115.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 123.14286041259766, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.8901925086975098, "rewards/meter/std": 0.22116033732891083, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8442056179046631, "rewards/repeat_soft/std": 0.10568331182003021, "rewards/judge_quality/mean": 0.6862499713897705, "rewards/judge_quality/std": 0.34221702814102173, "rewards/total_composite/mean": 0.7890921831130981, "rewards/total_composite/std": 0.32686787843704224, "reward": 0.7890921831130981, "reward_std": 0.3268679082393646, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08911658078432083, "sampling/sampling_logp_difference/max": 3.604128837585449, "sampling/importance_sampling_ratio/min": 0.027211138978600502, "sampling/importance_sampling_ratio/mean": 1.0137627124786377, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4051705151796341, "clip_ratio/low_mean": 0.010338345542550087, "clip_ratio/low_min": 0.010338345542550087, "clip_ratio/high_mean": 0.07043441757559776, "clip_ratio/high_max": 0.07043441757559776, "clip_ratio/region_mean": 0.08077276311814785, "reward_total_mean": 0.7890921831130981, "reward_meter_mean": 0.8901925086975098, "reward_meter_std": 0.22116033732891083, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8442056179046631, "reward_repeat_soft_std": 0.10568331182003021, "reward_judge_quality_mean": 0.6862499713897705, "reward_judge_quality_std": 0.34221702814102173, "reward_total_composite_mean": 0.7890921831130981, "reward_total_composite_std": 0.32686787843704224} {"timestamp_utc": "2026-04-13T05:26:07Z", "mode": "train", "global_step": 2824, "epoch": 0.2836765444500251, "loss": 0.04, "grad_norm": 12.012691497802734, "learning_rate": 1.4454545454545453e-06, "num_tokens": 5270036.0, "completions/mean_length": 56.0, "completions/min_length": 50.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.0, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.8872206211090088, "rewards/meter/std": 0.26458680629730225, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9741036891937256, "rewards/repeat_soft/std": 0.025439612567424774, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.19334925711154938, "rewards/total_composite/mean": 0.7872846126556396, "rewards/total_composite/std": 0.06634648889303207, "reward": 0.7872846126556396, "reward_std": 0.06634647399187088, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09137625992298126, "sampling/sampling_logp_difference/max": 1.540503740310669, "sampling/importance_sampling_ratio/min": 0.214273139834404, "sampling/importance_sampling_ratio/mean": 1.0141677856445312, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4381624571979046, "clip_ratio/low_mean": 0.021702964790165424, "clip_ratio/low_min": 0.021702964790165424, "clip_ratio/high_mean": 0.06101302895694971, "clip_ratio/high_max": 0.06101302895694971, "clip_ratio/region_mean": 0.08271599374711514, "reward_total_mean": 0.7872846126556396, "reward_meter_mean": 0.8872206211090088, "reward_meter_std": 0.26458680629730225, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9741036891937256, "reward_repeat_soft_std": 0.025439612567424774, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.19334925711154938, "reward_total_composite_mean": 0.7872846126556396, "reward_total_composite_std": 0.06634648889303207} {"timestamp_utc": "2026-04-13T05:26:18Z", "mode": "train", "global_step": 2825, "epoch": 0.2837769964841788, "loss": -0.085, "grad_norm": 1.5109916925430298, "learning_rate": 1.4424242424242426e-06, "num_tokens": 5271412.0, "completions/mean_length": 85.0, "completions/min_length": 20.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 24.000001907348633, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.8469936847686768, "rewards/meter/std": 0.3436232805252075, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9145979285240173, "rewards/repeat_soft/std": 0.07829728722572327, "rewards/judge_quality/mean": 0.45125001668930054, "rewards/judge_quality/std": 0.23381541669368744, "rewards/total_composite/mean": 0.724856972694397, "rewards/total_composite/std": 0.29895731806755066, "reward": 0.724856972694397, "reward_std": 0.29895731806755066, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08473846316337585, "sampling/sampling_logp_difference/max": 1.0328686237335205, "sampling/importance_sampling_ratio/min": 0.35598433017730713, "sampling/importance_sampling_ratio/mean": 1.0337601900100708, "sampling/importance_sampling_ratio/max": 1.8359010219573975, "entropy": 0.46907680854201317, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.07055555656552315, "clip_ratio/high_max": 0.07055555656552315, "clip_ratio/region_mean": 0.07055555656552315, "reward_total_mean": 0.724856972694397, "reward_meter_mean": 0.8469936847686768, "reward_meter_std": 0.3436232805252075, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9145979285240173, "reward_repeat_soft_std": 0.07829728722572327, "reward_judge_quality_mean": 0.45125001668930054, "reward_judge_quality_std": 0.23381541669368744, "reward_total_composite_mean": 0.724856972694397, "reward_total_composite_std": 0.29895731806755066} {"timestamp_utc": "2026-04-13T05:26:25Z", "mode": "train", "global_step": 2826, "epoch": 0.2838774485183325, "loss": -0.0109, "grad_norm": 4.566472053527832, "learning_rate": 1.4393939393939396e-06, "num_tokens": 5273829.0, "completions/mean_length": 116.125, "completions/min_length": 105.0, "completions/max_length": 127.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.125, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.9720326662063599, "rewards/meter/std": 0.005350543186068535, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.66303551197052, "rewards/repeat_soft/std": 0.0979895070195198, "rewards/judge_quality/mean": 0.2462499886751175, "rewards/judge_quality/std": 0.08348438143730164, "rewards/total_composite/mean": 0.7275932431221008, "rewards/total_composite/std": 0.028035176917910576, "reward": 0.7275932431221008, "reward_std": 0.028035158291459084, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.041115257889032364, "sampling/sampling_logp_difference/max": 2.114773750305176, "sampling/importance_sampling_ratio/min": 0.12066058814525604, "sampling/importance_sampling_ratio/mean": 1.005503535270691, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15871799923479557, "clip_ratio/low_mean": 0.017785534728318453, "clip_ratio/low_min": 0.017785534728318453, "clip_ratio/high_mean": 0.016755036544054747, "clip_ratio/high_max": 0.016755036544054747, "clip_ratio/region_mean": 0.0345405712723732, "reward_total_mean": 0.7275932431221008, "reward_meter_mean": 0.9720326662063599, "reward_meter_std": 0.005350543186068535, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.66303551197052, "reward_repeat_soft_std": 0.0979895070195198, "reward_judge_quality_mean": 0.2462499886751175, "reward_judge_quality_std": 0.08348438143730164, "reward_total_composite_mean": 0.7275932431221008, "reward_total_composite_std": 0.028035176917910576} {"timestamp_utc": "2026-04-13T05:26:32Z", "mode": "train", "global_step": 2827, "epoch": 0.2839779005524862, "loss": 0.0218, "grad_norm": 6.266737461090088, "learning_rate": 1.4363636363636365e-06, "num_tokens": 5275626.0, "completions/mean_length": 65.625, "completions/min_length": 62.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.625, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9315505027770996, "rewards/meter/std": 0.08887765556573868, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9242908954620361, "rewards/repeat_soft/std": 0.05657753720879555, "rewards/judge_quality/mean": 0.3462499976158142, "rewards/judge_quality/std": 0.10336308926343918, "rewards/total_composite/mean": 0.765501856803894, "rewards/total_composite/std": 0.04196227341890335, "reward": 0.765501856803894, "reward_std": 0.041962284594774246, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.058213189244270325, "sampling/sampling_logp_difference/max": 1.0328030586242676, "sampling/importance_sampling_ratio/min": 0.3930545449256897, "sampling/importance_sampling_ratio/mean": 1.0072232484817505, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3029076885432005, "clip_ratio/low_mean": 0.04220846178941429, "clip_ratio/low_min": 0.04220846178941429, "clip_ratio/high_mean": 0.011251434800215065, "clip_ratio/high_max": 0.011251434800215065, "clip_ratio/region_mean": 0.05345989658962935, "reward_total_mean": 0.765501856803894, "reward_meter_mean": 0.9315505027770996, "reward_meter_std": 0.08887765556573868, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9242908954620361, "reward_repeat_soft_std": 0.05657753720879555, "reward_judge_quality_mean": 0.3462499976158142, "reward_judge_quality_std": 0.10336308926343918, "reward_total_composite_mean": 0.765501856803894, "reward_total_composite_std": 0.04196227341890335} {"timestamp_utc": "2026-04-13T05:26:38Z", "mode": "train", "global_step": 2828, "epoch": 0.2840783525866399, "loss": 0.0098, "grad_norm": 19.838594436645508, "learning_rate": 1.4333333333333335e-06, "num_tokens": 5277137.0, "completions/mean_length": 28.875, "completions/min_length": 25.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.875, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.5734712481498718, "rewards/meter/std": 0.4356398284435272, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9457767009735107, "rewards/repeat_soft/std": 0.03811241686344147, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.6683897376060486, "rewards/total_composite/std": 0.2054402232170105, "reward": 0.6683897376060486, "reward_std": 0.2054402232170105, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10470551252365112, "sampling/sampling_logp_difference/max": 2.0148849487304688, "sampling/importance_sampling_ratio/min": 0.13333575427532196, "sampling/importance_sampling_ratio/mean": 1.0132122039794922, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6977154687047005, "clip_ratio/low_mean": 0.05390989314764738, "clip_ratio/low_min": 0.05390989314764738, "clip_ratio/high_mean": 0.05334087647497654, "clip_ratio/high_max": 0.05334087647497654, "clip_ratio/region_mean": 0.10725076962262392, "reward_total_mean": 0.6683897376060486, "reward_meter_mean": 0.5734712481498718, "reward_meter_std": 0.4356398284435272, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9457767009735107, "reward_repeat_soft_std": 0.03811241686344147, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.6683897376060486, "reward_total_composite_std": 0.2054402232170105} {"timestamp_utc": "2026-04-13T05:26:49Z", "mode": "train", "global_step": 2829, "epoch": 0.2841788046207936, "loss": -0.1222, "grad_norm": 1.556023359298706, "learning_rate": 1.4303030303030306e-06, "num_tokens": 5278694.0, "completions/mean_length": 167.625, "completions/min_length": 45.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 52.833335876464844, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.746484637260437, "rewards/meter/std": 0.4391646087169647, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.3720119297504425, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9572499990463257, "rewards/repeat_soft/std": 0.0418960265815258, "rewards/judge_quality/mean": 0.518750011920929, "rewards/judge_quality/std": 0.3677513897418976, "rewards/total_composite/mean": 0.6674603819847107, "rewards/total_composite/std": 0.41747546195983887, "reward": 0.6674603819847107, "reward_std": 0.41747546195983887, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10153838992118835, "sampling/sampling_logp_difference/max": 1.351304531097412, "sampling/importance_sampling_ratio/min": 0.25890231132507324, "sampling/importance_sampling_ratio/mean": 1.0085936784744263, "sampling/importance_sampling_ratio/max": 1.709192156791687, "entropy": 0.31504589319229126, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06482920236885548, "clip_ratio/high_max": 0.06482920236885548, "clip_ratio/region_mean": 0.06482920236885548, "reward_total_mean": 0.6674603819847107, "reward_meter_mean": 0.746484637260437, "reward_meter_std": 0.4391646087169647, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.3720119297504425, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9572499990463257, "reward_repeat_soft_std": 0.0418960265815258, "reward_judge_quality_mean": 0.518750011920929, "reward_judge_quality_std": 0.3677513897418976, "reward_total_composite_mean": 0.6674603819847107, "reward_total_composite_std": 0.41747546195983887} {"timestamp_utc": "2026-04-13T05:26:56Z", "mode": "train", "global_step": 2830, "epoch": 0.28427925665494724, "loss": 0.0477, "grad_norm": 10.097790718078613, "learning_rate": 1.4272727272727274e-06, "num_tokens": 5280557.0, "completions/mean_length": 58.875, "completions/min_length": 51.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.875, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9079073667526245, "rewards/meter/std": 0.14217537641525269, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.986664891242981, "rewards/repeat_soft/std": 0.021711044013500214, "rewards/judge_quality/mean": 0.45249998569488525, "rewards/judge_quality/std": 0.18100909888744354, "rewards/total_composite/mean": 0.7929748296737671, "rewards/total_composite/std": 0.08917219936847687, "reward": 0.7929748296737671, "reward_std": 0.08917217701673508, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11576014757156372, "sampling/sampling_logp_difference/max": 2.1355128288269043, "sampling/importance_sampling_ratio/min": 0.11818396300077438, "sampling/importance_sampling_ratio/mean": 1.0226737260818481, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6737100780010223, "clip_ratio/low_mean": 0.05587056977674365, "clip_ratio/low_min": 0.05587056977674365, "clip_ratio/high_mean": 0.058253816328942776, "clip_ratio/high_max": 0.058253816328942776, "clip_ratio/region_mean": 0.11412438610568643, "reward_total_mean": 0.7929748296737671, "reward_meter_mean": 0.9079073667526245, "reward_meter_std": 0.14217537641525269, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.986664891242981, "reward_repeat_soft_std": 0.021711044013500214, "reward_judge_quality_mean": 0.45249998569488525, "reward_judge_quality_std": 0.18100909888744354, "reward_total_composite_mean": 0.7929748296737671, "reward_total_composite_std": 0.08917219936847687} {"timestamp_utc": "2026-04-13T05:27:04Z", "mode": "train", "global_step": 2831, "epoch": 0.28437970868910095, "loss": 0.0144, "grad_norm": 6.839375019073486, "learning_rate": 1.4242424242424244e-06, "num_tokens": 5282969.0, "completions/mean_length": 128.5, "completions/min_length": 116.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.5, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.9861294627189636, "rewards/meter/std": 0.01876157894730568, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9120374917984009, "rewards/repeat_soft/std": 0.021230608224868774, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8109620213508606, "rewards/total_composite/std": 0.008689316920936108, "reward": 0.8109620213508606, "reward_std": 0.00868931133300066, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0724877417087555, "sampling/sampling_logp_difference/max": 2.1714067459106445, "sampling/importance_sampling_ratio/min": 0.11401711404323578, "sampling/importance_sampling_ratio/mean": 1.0070185661315918, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3283983804285526, "clip_ratio/low_mean": 0.014694574289023876, "clip_ratio/low_min": 0.014694574289023876, "clip_ratio/high_mean": 0.05349011532962322, "clip_ratio/high_max": 0.05349011532962322, "clip_ratio/region_mean": 0.0681846896186471, "reward_total_mean": 0.8109620213508606, "reward_meter_mean": 0.9861294627189636, "reward_meter_std": 0.01876157894730568, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9120374917984009, "reward_repeat_soft_std": 0.021230608224868774, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8109620213508606, "reward_total_composite_std": 0.008689316920936108} {"timestamp_utc": "2026-04-13T05:27:11Z", "mode": "train", "global_step": 2832, "epoch": 0.28448016072325466, "loss": 0.0361, "grad_norm": 7.584771633148193, "learning_rate": 1.4212121212121213e-06, "num_tokens": 5284843.0, "completions/mean_length": 73.25, "completions/min_length": 70.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.25, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9914790391921997, "rewards/meter/std": 0.0055784801952540874, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.957419753074646, "rewards/repeat_soft/std": 0.04181412607431412, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.8576575517654419, "rewards/total_composite/std": 0.06871796399354935, "reward": 0.8576575517654419, "reward_std": 0.06871795654296875, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09070242941379547, "sampling/sampling_logp_difference/max": 3.3575949668884277, "sampling/importance_sampling_ratio/min": 0.034818898886442184, "sampling/importance_sampling_ratio/mean": 1.0043336153030396, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49880627915263176, "clip_ratio/low_mean": 0.0405215653590858, "clip_ratio/low_min": 0.0405215653590858, "clip_ratio/high_mean": 0.010567514691501856, "clip_ratio/high_max": 0.010567514691501856, "clip_ratio/region_mean": 0.051089080050587654, "reward_total_mean": 0.8576575517654419, "reward_meter_mean": 0.9914790391921997, "reward_meter_std": 0.0055784801952540874, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.957419753074646, "reward_repeat_soft_std": 0.04181412607431412, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.8576575517654419, "reward_total_composite_std": 0.06871796399354935} {"timestamp_utc": "2026-04-13T05:27:18Z", "mode": "train", "global_step": 2833, "epoch": 0.2845806127574083, "loss": 0.0274, "grad_norm": 6.837702751159668, "learning_rate": 1.4181818181818183e-06, "num_tokens": 5287098.0, "completions/mean_length": 100.875, "completions/min_length": 96.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.875, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.9887696504592896, "rewards/meter/std": 0.00383979850448668, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8586245775222778, "rewards/repeat_soft/std": 0.05901992321014404, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.8443088531494141, "rewards/total_composite/std": 0.07225988805294037, "reward": 0.8443088531494141, "reward_std": 0.07225988060235977, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08178200572729111, "sampling/sampling_logp_difference/max": 3.94130802154541, "sampling/importance_sampling_ratio/min": 0.01942279189825058, "sampling/importance_sampling_ratio/mean": 0.9946683049201965, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.28887875750660896, "clip_ratio/low_mean": 0.057798401452600956, "clip_ratio/low_min": 0.057798401452600956, "clip_ratio/high_mean": 0.022375606931746006, "clip_ratio/high_max": 0.022375606931746006, "clip_ratio/region_mean": 0.08017400838434696, "reward_total_mean": 0.8443088531494141, "reward_meter_mean": 0.9887696504592896, "reward_meter_std": 0.00383979850448668, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8586245775222778, "reward_repeat_soft_std": 0.05901992321014404, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.8443088531494141, "reward_total_composite_std": 0.07225988805294037} {"timestamp_utc": "2026-04-13T05:27:26Z", "mode": "train", "global_step": 2834, "epoch": 0.284681064791562, "loss": 0.0053, "grad_norm": 10.16480541229248, "learning_rate": 1.4151515151515151e-06, "num_tokens": 5288765.0, "completions/mean_length": 45.375, "completions/min_length": 43.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.375, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.967783510684967, "rewards/meter/std": 0.00948020163923502, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9451189041137695, "rewards/repeat_soft/std": 0.030237726867198944, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8082644939422607, "rewards/total_composite/std": 0.007729531265795231, "reward": 0.8082644939422607, "reward_std": 0.007729535456746817, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09452573955059052, "sampling/sampling_logp_difference/max": 2.1675171852111816, "sampling/importance_sampling_ratio/min": 0.11446144431829453, "sampling/importance_sampling_ratio/mean": 0.999721884727478, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37944988161325455, "clip_ratio/low_mean": 0.05274209380149841, "clip_ratio/low_min": 0.05274209380149841, "clip_ratio/high_mean": 0.038917838828638196, "clip_ratio/high_max": 0.038917838828638196, "clip_ratio/region_mean": 0.09165993263013661, "reward_total_mean": 0.8082644939422607, "reward_meter_mean": 0.967783510684967, "reward_meter_std": 0.00948020163923502, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9451189041137695, "reward_repeat_soft_std": 0.030237726867198944, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8082644939422607, "reward_total_composite_std": 0.007729531265795231} {"timestamp_utc": "2026-04-13T05:27:33Z", "mode": "train", "global_step": 2835, "epoch": 0.28478151682571573, "loss": 0.0235, "grad_norm": 6.8882880210876465, "learning_rate": 1.4121212121212122e-06, "num_tokens": 5290490.0, "completions/mean_length": 65.625, "completions/min_length": 62.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.625, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9872580766677856, "rewards/meter/std": 0.008878618478775024, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8696298599243164, "rewards/repeat_soft/std": 0.05286271497607231, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8072291016578674, "rewards/total_composite/std": 0.00564031396061182, "reward": 0.8072291016578674, "reward_std": 0.005640329793095589, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06643764674663544, "sampling/sampling_logp_difference/max": 1.2132453918457031, "sampling/importance_sampling_ratio/min": 0.2972310781478882, "sampling/importance_sampling_ratio/mean": 0.9992104768753052, "sampling/importance_sampling_ratio/max": 1.9387096166610718, "entropy": 0.3565882034599781, "clip_ratio/low_mean": 0.05141779547557235, "clip_ratio/low_min": 0.05141779547557235, "clip_ratio/high_mean": 0.03498759353533387, "clip_ratio/high_max": 0.03498759353533387, "clip_ratio/region_mean": 0.08640538901090622, "reward_total_mean": 0.8072291016578674, "reward_meter_mean": 0.9872580766677856, "reward_meter_std": 0.008878618478775024, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8696298599243164, "reward_repeat_soft_std": 0.05286271497607231, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8072291016578674, "reward_total_composite_std": 0.00564031396061182} {"timestamp_utc": "2026-04-13T05:27:39Z", "mode": "train", "global_step": 2836, "epoch": 0.28488196885986944, "loss": -0.0133, "grad_norm": 9.993071556091309, "learning_rate": 1.409090909090909e-06, "num_tokens": 5292067.0, "completions/mean_length": 42.125, "completions/min_length": 37.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.125, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.7588485479354858, "rewards/meter/std": 0.3831457495689392, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9413135051727295, "rewards/repeat_soft/std": 0.029470685869455338, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.7888631820678711, "rewards/total_composite/std": 0.1359209418296814, "reward": 0.7888631820678711, "reward_std": 0.1359209567308426, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09943793714046478, "sampling/sampling_logp_difference/max": 1.3062043190002441, "sampling/importance_sampling_ratio/min": 0.2708461582660675, "sampling/importance_sampling_ratio/mean": 0.9996823072433472, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44211703538894653, "clip_ratio/low_mean": 0.04501689039170742, "clip_ratio/low_min": 0.04501689039170742, "clip_ratio/high_mean": 0.07739320420660079, "clip_ratio/high_max": 0.07739320420660079, "clip_ratio/region_mean": 0.1224100945983082, "reward_total_mean": 0.7888631820678711, "reward_meter_mean": 0.7588485479354858, "reward_meter_std": 0.3831457495689392, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9413135051727295, "reward_repeat_soft_std": 0.029470685869455338, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.7888631820678711, "reward_total_composite_std": 0.1359209418296814} {"timestamp_utc": "2026-04-13T05:27:52Z", "mode": "train", "global_step": 2837, "epoch": 0.2849824208940231, "loss": -0.1732, "grad_norm": 0.890784740447998, "learning_rate": 1.406060606060606e-06, "num_tokens": 5293999.0, "completions/mean_length": 126.5, "completions/min_length": 68.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 71.42857360839844, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.8629764318466187, "rewards/meter/std": 0.34871259331703186, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.6718473434448242, "rewards/repeat_soft/std": 0.17972426116466522, "rewards/judge_quality/mean": 0.3787499964237213, "rewards/judge_quality/std": 0.1537565290927887, "rewards/total_composite/mean": 0.687899112701416, "rewards/total_composite/std": 0.27814674377441406, "reward": 0.687899112701416, "reward_std": 0.27814674377441406, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.044068675488233566, "sampling/sampling_logp_difference/max": 0.8771724700927734, "sampling/importance_sampling_ratio/min": 0.41595736145973206, "sampling/importance_sampling_ratio/mean": 1.0013729333877563, "sampling/importance_sampling_ratio/max": 1.9013766050338745, "entropy": 0.1723195631057024, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.04363928979728371, "clip_ratio/high_max": 0.04363928979728371, "clip_ratio/region_mean": 0.04363928979728371, "reward_total_mean": 0.687899112701416, "reward_meter_mean": 0.8629764318466187, "reward_meter_std": 0.34871259331703186, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.6718473434448242, "reward_repeat_soft_std": 0.17972426116466522, "reward_judge_quality_mean": 0.3787499964237213, "reward_judge_quality_std": 0.1537565290927887, "reward_total_composite_mean": 0.687899112701416, "reward_total_composite_std": 0.27814674377441406} {"timestamp_utc": "2026-04-13T05:27:59Z", "mode": "train", "global_step": 2838, "epoch": 0.2850828729281768, "loss": -0.0272, "grad_norm": 5.143437385559082, "learning_rate": 1.403030303030303e-06, "num_tokens": 5296850.0, "completions/mean_length": 175.375, "completions/min_length": 157.0, "completions/max_length": 197.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 175.375, "completions/min_terminated_length": 157.0, "completions/max_terminated_length": 197.0, "rewards/meter/mean": 0.9927654266357422, "rewards/meter/std": 0.004005319904536009, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8518975973129272, "rewards/repeat_soft/std": 0.033166687935590744, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8004342317581177, "rewards/total_composite/std": 0.012648564763367176, "reward": 0.8004342317581177, "reward_std": 0.01264855545014143, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0750369057059288, "sampling/sampling_logp_difference/max": 2.182533025741577, "sampling/importance_sampling_ratio/min": 0.11275555938482285, "sampling/importance_sampling_ratio/mean": 1.0048354864120483, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.38813208043575287, "clip_ratio/low_mean": 0.011891459114849567, "clip_ratio/low_min": 0.011891459114849567, "clip_ratio/high_mean": 0.052042046561837196, "clip_ratio/high_max": 0.052042046561837196, "clip_ratio/region_mean": 0.06393350567668676, "reward_total_mean": 0.8004342317581177, "reward_meter_mean": 0.9927654266357422, "reward_meter_std": 0.004005319904536009, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8518975973129272, "reward_repeat_soft_std": 0.033166687935590744, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8004342317581177, "reward_total_composite_std": 0.012648564763367176} {"timestamp_utc": "2026-04-13T05:28:06Z", "mode": "train", "global_step": 2839, "epoch": 0.2851833249623305, "loss": -0.0225, "grad_norm": 8.632295608520508, "learning_rate": 1.4000000000000001e-06, "num_tokens": 5298566.0, "completions/mean_length": 55.5, "completions/min_length": 50.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.5, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.979201078414917, "rewards/meter/std": 0.022991111502051353, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9671087265014648, "rewards/repeat_soft/std": 0.041028331965208054, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8189762830734253, "rewards/total_composite/std": 0.010412439703941345, "reward": 0.8189762830734253, "reward_std": 0.010412439703941345, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09485514461994171, "sampling/sampling_logp_difference/max": 1.0319567918777466, "sampling/importance_sampling_ratio/min": 0.35630905628204346, "sampling/importance_sampling_ratio/mean": 1.0009114742279053, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4677596017718315, "clip_ratio/low_mean": 0.02773336647078395, "clip_ratio/low_min": 0.02773336647078395, "clip_ratio/high_mean": 0.05840445403009653, "clip_ratio/high_max": 0.05840445403009653, "clip_ratio/region_mean": 0.08613782050088048, "reward_total_mean": 0.8189762830734253, "reward_meter_mean": 0.979201078414917, "reward_meter_std": 0.022991111502051353, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9671087265014648, "reward_repeat_soft_std": 0.041028331965208054, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8189762830734253, "reward_total_composite_std": 0.010412439703941345} {"timestamp_utc": "2026-04-13T05:28:18Z", "mode": "train", "global_step": 2840, "epoch": 0.28528377699648416, "loss": -0.0773, "grad_norm": 2.564250946044922, "learning_rate": 1.3969696969696972e-06, "num_tokens": 5299972.0, "completions/mean_length": 88.75, "completions/min_length": 24.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 28.285715103149414, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.8259468674659729, "rewards/meter/std": 0.29008692502975464, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.949100136756897, "rewards/repeat_soft/std": 0.030872376635670662, "rewards/judge_quality/mean": 0.3137499690055847, "rewards/judge_quality/std": 0.13845448195934296, "rewards/total_composite/mean": 0.6491140127182007, "rewards/total_composite/std": 0.28333014249801636, "reward": 0.6491140127182007, "reward_std": 0.28333014249801636, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11183267831802368, "sampling/sampling_logp_difference/max": 2.3723978996276855, "sampling/importance_sampling_ratio/min": 0.09325683116912842, "sampling/importance_sampling_ratio/mean": 0.9813154339790344, "sampling/importance_sampling_ratio/max": 1.8843412399291992, "entropy": 0.3921792134642601, "clip_ratio/low_mean": 0.007352941203862429, "clip_ratio/low_min": 0.007352941203862429, "clip_ratio/high_mean": 0.09502047766000032, "clip_ratio/high_max": 0.09502047766000032, "clip_ratio/region_mean": 0.10237341886386275, "reward_total_mean": 0.6491140127182007, "reward_meter_mean": 0.8259468674659729, "reward_meter_std": 0.29008692502975464, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.949100136756897, "reward_repeat_soft_std": 0.030872376635670662, "reward_judge_quality_mean": 0.3137499690055847, "reward_judge_quality_std": 0.13845448195934296, "reward_total_composite_mean": 0.6491140127182007, "reward_total_composite_std": 0.28333014249801636} {"timestamp_utc": "2026-04-13T05:28:29Z", "mode": "train", "global_step": 2841, "epoch": 0.28538422903063787, "loss": -0.2464, "grad_norm": 1.4458457231521606, "learning_rate": 1.3939393939393942e-06, "num_tokens": 5302792.0, "completions/mean_length": 223.5, "completions/min_length": 171.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 182.2857208251953, "completions/min_terminated_length": 171.0, "completions/max_terminated_length": 209.0, "rewards/meter/mean": 0.9904106259346008, "rewards/meter/std": 0.002413436071947217, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.75345778465271, "rewards/repeat_soft/std": 0.07233354449272156, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.6934150457382202, "rewards/total_composite/std": 0.2803165018558502, "reward": 0.6934150457382202, "reward_std": 0.2803165018558502, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07285480946302414, "sampling/sampling_logp_difference/max": 2.020775318145752, "sampling/importance_sampling_ratio/min": 0.13255265355110168, "sampling/importance_sampling_ratio/mean": 0.9973210096359253, "sampling/importance_sampling_ratio/max": 1.8458818197250366, "entropy": 0.3396788351237774, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06727159908041358, "clip_ratio/high_max": 0.06727159908041358, "clip_ratio/region_mean": 0.06727159908041358, "reward_total_mean": 0.6934150457382202, "reward_meter_mean": 0.9904106259346008, "reward_meter_std": 0.002413436071947217, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.75345778465271, "reward_repeat_soft_std": 0.07233354449272156, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.6934150457382202, "reward_total_composite_std": 0.2803165018558502} {"timestamp_utc": "2026-04-13T05:28:35Z", "mode": "train", "global_step": 2842, "epoch": 0.2854846810647916, "loss": -0.0046, "grad_norm": 13.27961540222168, "learning_rate": 1.390909090909091e-06, "num_tokens": 5304208.0, "completions/mean_length": 27.0, "completions/min_length": 25.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.0, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.9877179861068726, "rewards/meter/std": 0.004658795893192291, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9368294477462769, "rewards/repeat_soft/std": 0.03445558249950409, "rewards/judge_quality/mean": 0.6850000023841858, "rewards/judge_quality/std": 0.2512255907058716, "rewards/total_composite/mean": 0.8936560153961182, "rewards/total_composite/std": 0.0730518251657486, "reward": 0.8936560153961182, "reward_std": 0.0730518102645874, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0935114324092865, "sampling/sampling_logp_difference/max": 1.472146987915039, "sampling/importance_sampling_ratio/min": 0.2294323742389679, "sampling/importance_sampling_ratio/mean": 1.0195927619934082, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4947265349328518, "clip_ratio/low_mean": 0.06189560517668724, "clip_ratio/low_min": 0.06189560517668724, "clip_ratio/high_mean": 0.028199233580380678, "clip_ratio/high_max": 0.028199233580380678, "clip_ratio/region_mean": 0.09009483875706792, "reward_total_mean": 0.8936560153961182, "reward_meter_mean": 0.9877179861068726, "reward_meter_std": 0.004658795893192291, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9368294477462769, "reward_repeat_soft_std": 0.03445558249950409, "reward_judge_quality_mean": 0.6850000023841858, "reward_judge_quality_std": 0.2512255907058716, "reward_total_composite_mean": 0.8936560153961182, "reward_total_composite_std": 0.0730518251657486} {"timestamp_utc": "2026-04-13T05:28:42Z", "mode": "train", "global_step": 2843, "epoch": 0.28558513309894523, "loss": 0.0595, "grad_norm": 12.593438148498535, "learning_rate": 1.3878787878787881e-06, "num_tokens": 5305955.0, "completions/mean_length": 48.375, "completions/min_length": 43.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.375, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9841168522834778, "rewards/meter/std": 0.012700811959803104, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9597185850143433, "rewards/repeat_soft/std": 0.0457424558699131, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.9085744023323059, "rewards/total_composite/std": 0.08413958549499512, "reward": 0.9085744023323059, "reward_std": 0.08413958549499512, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09818227589130402, "sampling/sampling_logp_difference/max": 2.243000030517578, "sampling/importance_sampling_ratio/min": 0.10613960772752762, "sampling/importance_sampling_ratio/mean": 1.0064641237258911, "sampling/importance_sampling_ratio/max": 1.8193538188934326, "entropy": 0.5144626423716545, "clip_ratio/low_mean": 0.04852697905153036, "clip_ratio/low_min": 0.04852697905153036, "clip_ratio/high_mean": 0.06671146210283041, "clip_ratio/high_max": 0.06671146210283041, "clip_ratio/region_mean": 0.11523844115436077, "reward_total_mean": 0.9085744023323059, "reward_meter_mean": 0.9841168522834778, "reward_meter_std": 0.012700811959803104, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9597185850143433, "reward_repeat_soft_std": 0.0457424558699131, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.9085744023323059, "reward_total_composite_std": 0.08413958549499512} {"timestamp_utc": "2026-04-13T05:28:50Z", "mode": "train", "global_step": 2844, "epoch": 0.28568558513309894, "loss": 0.0579, "grad_norm": 4.936332702636719, "learning_rate": 1.384848484848485e-06, "num_tokens": 5308740.0, "completions/mean_length": 162.125, "completions/min_length": 134.0, "completions/max_length": 179.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 162.125, "completions/min_terminated_length": 134.0, "completions/max_terminated_length": 179.0, "rewards/meter/mean": 0.9839963912963867, "rewards/meter/std": 0.01249384880065918, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7822434902191162, "rewards/repeat_soft/std": 0.05795292556285858, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.7741477489471436, "rewards/total_composite/std": 0.030782807618379593, "reward": 0.7741477489471436, "reward_std": 0.030782822519540787, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0673295110464096, "sampling/sampling_logp_difference/max": 1.9515984058380127, "sampling/importance_sampling_ratio/min": 0.14204683899879456, "sampling/importance_sampling_ratio/mean": 1.004622220993042, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3169238958507776, "clip_ratio/low_mean": 0.02907061530277133, "clip_ratio/low_min": 0.02907061530277133, "clip_ratio/high_mean": 0.0354961296543479, "clip_ratio/high_max": 0.0354961296543479, "clip_ratio/region_mean": 0.06456674495711923, "reward_total_mean": 0.7741477489471436, "reward_meter_mean": 0.9839963912963867, "reward_meter_std": 0.01249384880065918, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7822434902191162, "reward_repeat_soft_std": 0.05795292556285858, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.7741477489471436, "reward_total_composite_std": 0.030782807618379593} {"timestamp_utc": "2026-04-13T05:28:57Z", "mode": "train", "global_step": 2845, "epoch": 0.28578603716725265, "loss": -0.0077, "grad_norm": 9.495923042297363, "learning_rate": 1.381818181818182e-06, "num_tokens": 5310544.0, "completions/mean_length": 57.5, "completions/min_length": 50.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.5, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9925948977470398, "rewards/meter/std": 0.0024093538522720337, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9607142806053162, "rewards/repeat_soft/std": 0.03884856402873993, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.25587037205696106, "rewards/total_composite/mean": 0.8761141300201416, "rewards/total_composite/std": 0.07789348065853119, "reward": 0.8761141300201416, "reward_std": 0.07789348065853119, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09963849186897278, "sampling/sampling_logp_difference/max": 1.4679213762283325, "sampling/importance_sampling_ratio/min": 0.23040391504764557, "sampling/importance_sampling_ratio/mean": 1.0195775032043457, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.41170570626854897, "clip_ratio/low_mean": 0.0758857810869813, "clip_ratio/low_min": 0.0758857810869813, "clip_ratio/high_mean": 0.039944594725966454, "clip_ratio/high_max": 0.039944594725966454, "clip_ratio/region_mean": 0.11583037581294775, "reward_total_mean": 0.8761141300201416, "reward_meter_mean": 0.9925948977470398, "reward_meter_std": 0.0024093538522720337, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9607142806053162, "reward_repeat_soft_std": 0.03884856402873993, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.25587037205696106, "reward_total_composite_mean": 0.8761141300201416, "reward_total_composite_std": 0.07789348065853119} {"timestamp_utc": "2026-04-13T05:29:03Z", "mode": "train", "global_step": 2846, "epoch": 0.2858864892014063, "loss": 0.0247, "grad_norm": 9.561141967773438, "learning_rate": 1.3787878787878788e-06, "num_tokens": 5312265.0, "completions/mean_length": 53.125, "completions/min_length": 50.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.125, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9859621524810791, "rewards/meter/std": 0.0060521336272358894, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9544419646263123, "rewards/repeat_soft/std": 0.05062384903430939, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.8196271657943726, "rewards/total_composite/std": 0.006590151693671942, "reward": 0.8196271657943726, "reward_std": 0.006590154021978378, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08169224858283997, "sampling/sampling_logp_difference/max": 2.203892707824707, "sampling/importance_sampling_ratio/min": 0.11037266999483109, "sampling/importance_sampling_ratio/mean": 0.987549364566803, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3412339985370636, "clip_ratio/low_mean": 0.009531400166451931, "clip_ratio/low_min": 0.009531400166451931, "clip_ratio/high_mean": 0.05343438498675823, "clip_ratio/high_max": 0.05343438498675823, "clip_ratio/region_mean": 0.06296578515321016, "reward_total_mean": 0.8196271657943726, "reward_meter_mean": 0.9859621524810791, "reward_meter_std": 0.0060521336272358894, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9544419646263123, "reward_repeat_soft_std": 0.05062384903430939, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.8196271657943726, "reward_total_composite_std": 0.006590151693671942} {"timestamp_utc": "2026-04-13T05:29:09Z", "mode": "train", "global_step": 2847, "epoch": 0.28598694123556, "loss": -0.0002, "grad_norm": 12.408912658691406, "learning_rate": 1.3757575757575759e-06, "num_tokens": 5313706.0, "completions/mean_length": 29.125, "completions/min_length": 26.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.125, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9834722280502319, "rewards/meter/std": 0.014400674030184746, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9612975120544434, "rewards/repeat_soft/std": 0.0034009767696261406, "rewards/judge_quality/mean": 0.44999998807907104, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.823692262172699, "rewards/total_composite/std": 0.006383468862622976, "reward": 0.823692262172699, "reward_std": 0.006383486557751894, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09339429438114166, "sampling/sampling_logp_difference/max": 1.1393795013427734, "sampling/importance_sampling_ratio/min": 0.3200175166130066, "sampling/importance_sampling_ratio/mean": 1.011065125465393, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5104278437793255, "clip_ratio/low_mean": 0.036481901071965694, "clip_ratio/low_min": 0.036481901071965694, "clip_ratio/high_mean": 0.07813536282628775, "clip_ratio/high_max": 0.07813536282628775, "clip_ratio/region_mean": 0.11461726389825344, "reward_total_mean": 0.823692262172699, "reward_meter_mean": 0.9834722280502319, "reward_meter_std": 0.014400674030184746, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9612975120544434, "reward_repeat_soft_std": 0.0034009767696261406, "reward_judge_quality_mean": 0.44999998807907104, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.823692262172699, "reward_total_composite_std": 0.006383468862622976} {"timestamp_utc": "2026-04-13T05:29:15Z", "mode": "train", "global_step": 2848, "epoch": 0.2860873932697137, "loss": -0.0012, "grad_norm": 14.85158634185791, "learning_rate": 1.3727272727272727e-06, "num_tokens": 5315316.0, "completions/mean_length": 34.25, "completions/min_length": 31.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.25, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.9615630507469177, "rewards/meter/std": 0.05758388713002205, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9604166746139526, "rewards/repeat_soft/std": 0.00589255103841424, "rewards/judge_quality/mean": 0.44624999165534973, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8126200437545776, "rewards/total_composite/std": 0.028969228267669678, "reward": 0.8126200437545776, "reward_std": 0.02896924316883087, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11534656584262848, "sampling/sampling_logp_difference/max": 1.442718505859375, "sampling/importance_sampling_ratio/min": 0.2362845540046692, "sampling/importance_sampling_ratio/mean": 1.01020085811615, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4341391660273075, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/high_mean": 0.09450027626007795, "clip_ratio/high_max": 0.09450027626007795, "clip_ratio/region_mean": 0.09828815516084433, "reward_total_mean": 0.8126200437545776, "reward_meter_mean": 0.9615630507469177, "reward_meter_std": 0.05758388713002205, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9604166746139526, "reward_repeat_soft_std": 0.00589255103841424, "reward_judge_quality_mean": 0.44624999165534973, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8126200437545776, "reward_total_composite_std": 0.028969228267669678} {"timestamp_utc": "2026-04-13T05:29:22Z", "mode": "train", "global_step": 2849, "epoch": 0.2861878453038674, "loss": 0.0174, "grad_norm": 6.704261779785156, "learning_rate": 1.3696969696969697e-06, "num_tokens": 5317205.0, "completions/mean_length": 65.125, "completions/min_length": 61.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.125, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9848941564559937, "rewards/meter/std": 0.004897789563983679, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8893701434135437, "rewards/repeat_soft/std": 0.025655239820480347, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8092644214630127, "rewards/total_composite/std": 0.004611509386450052, "reward": 0.8092644214630127, "reward_std": 0.004611523821949959, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07565395534038544, "sampling/sampling_logp_difference/max": 1.6153748035430908, "sampling/importance_sampling_ratio/min": 0.19881613552570343, "sampling/importance_sampling_ratio/mean": 0.9870799779891968, "sampling/importance_sampling_ratio/max": 1.8675650358200073, "entropy": 0.30192065238952637, "clip_ratio/low_mean": 0.03257449180819094, "clip_ratio/low_min": 0.03257449180819094, "clip_ratio/high_mean": 0.030506933573633432, "clip_ratio/high_max": 0.030506933573633432, "clip_ratio/region_mean": 0.06308142538182437, "reward_total_mean": 0.8092644214630127, "reward_meter_mean": 0.9848941564559937, "reward_meter_std": 0.004897789563983679, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8893701434135437, "reward_repeat_soft_std": 0.025655239820480347, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8092644214630127, "reward_total_composite_std": 0.004611509386450052} {"timestamp_utc": "2026-04-13T05:29:28Z", "mode": "train", "global_step": 2850, "epoch": 0.2862882973380211, "loss": 0.0164, "grad_norm": 7.108034610748291, "learning_rate": 1.3666666666666668e-06, "num_tokens": 5319161.0, "completions/mean_length": 84.5, "completions/min_length": 80.0, "completions/max_length": 89.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.5, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 89.0, "rewards/meter/mean": 0.9806355237960815, "rewards/meter/std": 0.016002824530005455, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9685685634613037, "rewards/repeat_soft/std": 0.013144463300704956, "rewards/judge_quality/mean": 0.5774999856948853, "rewards/judge_quality/std": 0.22461079061031342, "rewards/total_composite/mean": 0.8613928556442261, "rewards/total_composite/std": 0.06648458540439606, "reward": 0.8613928556442261, "reward_std": 0.06648459285497665, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07108103483915329, "sampling/sampling_logp_difference/max": 1.2321064472198486, "sampling/importance_sampling_ratio/min": 0.2916775345802307, "sampling/importance_sampling_ratio/mean": 1.0024582147598267, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3198029324412346, "clip_ratio/low_mean": 0.05342549178749323, "clip_ratio/low_min": 0.05342549178749323, "clip_ratio/high_mean": 0.025523244868963957, "clip_ratio/high_max": 0.025523244868963957, "clip_ratio/region_mean": 0.07894873665645719, "reward_total_mean": 0.8613928556442261, "reward_meter_mean": 0.9806355237960815, "reward_meter_std": 0.016002824530005455, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9685685634613037, "reward_repeat_soft_std": 0.013144463300704956, "reward_judge_quality_mean": 0.5774999856948853, "reward_judge_quality_std": 0.22461079061031342, "reward_total_composite_mean": 0.8613928556442261, "reward_total_composite_std": 0.06648458540439606} {"timestamp_utc": "2026-04-13T05:30:22Z", "mode": "eval", "global_step": 2850, "epoch": 0.2862882973380211, "eval_loss": NaN, "eval_runtime": 53.3929, "eval_samples_per_second": 1.498, "eval_steps_per_second": 0.187, "eval_num_tokens": 5319161.0, "eval_completions/mean_length": 111.1125, "eval_completions/min_length": 40.2, "eval_completions/max_length": 234.4, "eval_completions/clipped_ratio": 0.0375, "eval_completions/mean_terminated_length": 95.4702392578125, "eval_completions/min_terminated_length": 40.2, "eval_completions/max_terminated_length": 164.5, "eval_rewards/meter/mean": 0.930886173248291, "eval_rewards/meter/std": 0.10802524178288878, "eval_rewards/count_adherence/mean": 0.9893750011920929, "eval_rewards/count_adherence/std": 0.025168102979660035, "eval_rewards/hard_gate/mean": 0.9625, "eval_rewards/hard_gate/std": 0.0816463440656662, "eval_rewards/repeat_soft/mean": 0.8986733019351959, "eval_rewards/repeat_soft/std": 0.08918801695108414, "eval_rewards/judge_quality/mean": 0.41912499368190764, "eval_rewards/judge_quality/std": 0.13098689783364534, "eval_rewards/total_composite/mean": 0.7639388024806977, "eval_rewards/total_composite/std": 0.11024350207298994, "eval_reward": 0.7639388024806977, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.03788834977895021, "eval_sampling/sampling_logp_difference/max": 0.9031846523284912, "eval_sampling/importance_sampling_ratio/min": 0.4192597150802612, "eval_sampling/importance_sampling_ratio/mean": 1.0074296832084655, "eval_sampling/importance_sampling_ratio/max": 1.3758219361305237, "eval_entropy": 0.3761892557144165, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7639388024806977, "eval_reward_meter_mean": 0.930886173248291, "eval_reward_meter_std": 0.10802524178288878, "eval_reward_count_adherence_mean": 0.9893750011920929, "eval_reward_count_adherence_std": 0.025168102979660035, "eval_reward_hard_gate_mean": 0.9625, "eval_reward_hard_gate_std": 0.0816463440656662, "eval_reward_repeat_soft_mean": 0.8986733019351959, "eval_reward_repeat_soft_std": 0.08918801695108414, "eval_reward_judge_quality_mean": 0.41912499368190764, "eval_reward_judge_quality_std": 0.13098689783364534, "eval_reward_total_composite_mean": 0.7639388024806977, "eval_reward_total_composite_std": 0.11024350207298994} {"timestamp_utc": "2026-04-13T05:30:31Z", "mode": "train", "global_step": 2851, "epoch": 0.2863887493721748, "loss": 0.0106, "grad_norm": 7.85559606552124, "learning_rate": 1.3636363636363636e-06, "num_tokens": 5321009.0, "completions/mean_length": 66.0, "completions/min_length": 64.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.980515718460083, "rewards/meter/std": 0.017091121524572372, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8912161588668823, "rewards/repeat_soft/std": 0.05353382229804993, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.8288536667823792, "rewards/total_composite/std": 0.0454617440700531, "reward": 0.8288536667823792, "reward_std": 0.04546172916889191, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08467335253953934, "sampling/sampling_logp_difference/max": 2.86075496673584, "sampling/importance_sampling_ratio/min": 0.0572255402803421, "sampling/importance_sampling_ratio/mean": 0.9903257489204407, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.35532021149992943, "clip_ratio/low_mean": 0.03780597052536905, "clip_ratio/low_min": 0.03780597052536905, "clip_ratio/high_mean": 0.020210597664117813, "clip_ratio/high_max": 0.020210597664117813, "clip_ratio/region_mean": 0.05801656818948686, "reward_total_mean": 0.8288536667823792, "reward_meter_mean": 0.980515718460083, "reward_meter_std": 0.017091121524572372, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8912161588668823, "reward_repeat_soft_std": 0.05353382229804993, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.8288536667823792, "reward_total_composite_std": 0.0454617440700531} {"timestamp_utc": "2026-04-13T05:30:38Z", "mode": "train", "global_step": 2852, "epoch": 0.2864892014063285, "loss": -0.0525, "grad_norm": 6.387120723724365, "learning_rate": 1.3606060606060607e-06, "num_tokens": 5322990.0, "completions/mean_length": 86.625, "completions/min_length": 73.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.625, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.9932830929756165, "rewards/meter/std": 0.0024402744602411985, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.17817415297031403, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8824121952056885, "rewards/repeat_soft/std": 0.07043735682964325, "rewards/judge_quality/mean": 0.47749996185302734, "rewards/judge_quality/std": 0.16263456642627716, "rewards/total_composite/mean": 0.8034685850143433, "rewards/total_composite/std": 0.06328845024108887, "reward": 0.8034685850143433, "reward_std": 0.06328845024108887, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.069608673453331, "sampling/sampling_logp_difference/max": 1.3030152320861816, "sampling/importance_sampling_ratio/min": 0.2717112898826599, "sampling/importance_sampling_ratio/mean": 1.004765272140503, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.32547565177083015, "clip_ratio/low_mean": 0.047705522272735834, "clip_ratio/low_min": 0.047705522272735834, "clip_ratio/high_mean": 0.02385852299630642, "clip_ratio/high_max": 0.02385852299630642, "clip_ratio/region_mean": 0.07156404526904225, "reward_total_mean": 0.8034685850143433, "reward_meter_mean": 0.9932830929756165, "reward_meter_std": 0.0024402744602411985, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.17817415297031403, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8824121952056885, "reward_repeat_soft_std": 0.07043735682964325, "reward_judge_quality_mean": 0.47749996185302734, "reward_judge_quality_std": 0.16263456642627716, "reward_total_composite_mean": 0.8034685850143433, "reward_total_composite_std": 0.06328845024108887} {"timestamp_utc": "2026-04-13T05:30:46Z", "mode": "train", "global_step": 2853, "epoch": 0.28658965344048215, "loss": 0.0454, "grad_norm": 7.968369483947754, "learning_rate": 1.357575757575758e-06, "num_tokens": 5325431.0, "completions/mean_length": 107.125, "completions/min_length": 93.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.125, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.9391993284225464, "rewards/meter/std": 0.06247522309422493, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.08625820279121399, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8482813835144043, "rewards/repeat_soft/std": 0.08004570007324219, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.7485928535461426, "rewards/total_composite/std": 0.045363038778305054, "reward": 0.7485928535461426, "reward_std": 0.04536304622888565, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10741600394248962, "sampling/sampling_logp_difference/max": 2.9033265113830566, "sampling/importance_sampling_ratio/min": 0.054840486496686935, "sampling/importance_sampling_ratio/mean": 0.9940061569213867, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.340984333306551, "clip_ratio/low_mean": 0.034243905916810036, "clip_ratio/low_min": 0.034243905916810036, "clip_ratio/high_mean": 0.06495850626379251, "clip_ratio/high_max": 0.06495850626379251, "clip_ratio/region_mean": 0.09920241218060255, "reward_total_mean": 0.7485928535461426, "reward_meter_mean": 0.9391993284225464, "reward_meter_std": 0.06247522309422493, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.08625820279121399, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8482813835144043, "reward_repeat_soft_std": 0.08004570007324219, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.7485928535461426, "reward_total_composite_std": 0.045363038778305054} {"timestamp_utc": "2026-04-13T05:30:52Z", "mode": "train", "global_step": 2854, "epoch": 0.28669010547463586, "loss": -0.0395, "grad_norm": 13.551575660705566, "learning_rate": 1.3545454545454547e-06, "num_tokens": 5326906.0, "completions/mean_length": 31.375, "completions/min_length": 23.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.375, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.994213342666626, "rewards/meter/std": 0.002642601728439331, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.824146032333374, "rewards/total_composite/std": 0.004561820533126593, "reward": 0.824146032333374, "reward_std": 0.0045618158765137196, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11093311756849289, "sampling/sampling_logp_difference/max": 1.6511931419372559, "sampling/importance_sampling_ratio/min": 0.1918209046125412, "sampling/importance_sampling_ratio/mean": 0.9964478611946106, "sampling/importance_sampling_ratio/max": 1.955630898475647, "entropy": 0.564183808863163, "clip_ratio/low_mean": 0.03597460803575814, "clip_ratio/low_min": 0.03597460803575814, "clip_ratio/high_mean": 0.06000628275796771, "clip_ratio/high_max": 0.06000628275796771, "clip_ratio/region_mean": 0.09598089079372585, "reward_total_mean": 0.824146032333374, "reward_meter_mean": 0.994213342666626, "reward_meter_std": 0.002642601728439331, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.824146032333374, "reward_total_composite_std": 0.004561820533126593} {"timestamp_utc": "2026-04-13T05:31:04Z", "mode": "train", "global_step": 2855, "epoch": 0.28679055750878957, "loss": 0.0239, "grad_norm": 15.454448699951172, "learning_rate": 1.3515151515151518e-06, "num_tokens": 5328238.0, "completions/mean_length": 27.5, "completions/min_length": 23.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.5, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.6614331603050232, "rewards/meter/std": 0.3943292796611786, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9415397047996521, "rewards/repeat_soft/std": 0.05928464233875275, "rewards/judge_quality/mean": 0.3462499976158142, "rewards/judge_quality/std": 0.10336308926343918, "rewards/total_composite/mean": 0.6456738710403442, "rewards/total_composite/std": 0.1671849638223648, "reward": 0.6456738710403442, "reward_std": 0.167184978723526, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08313034474849701, "sampling/sampling_logp_difference/max": 1.0199661254882812, "sampling/importance_sampling_ratio/min": 0.48472198843955994, "sampling/importance_sampling_ratio/mean": 0.9998823404312134, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5306403636932373, "clip_ratio/low_mean": 0.04890804644674063, "clip_ratio/low_min": 0.04890804644674063, "clip_ratio/high_mean": 0.07047101622447371, "clip_ratio/high_max": 0.07047101622447371, "clip_ratio/region_mean": 0.11937906267121434, "reward_total_mean": 0.6456738710403442, "reward_meter_mean": 0.6614331603050232, "reward_meter_std": 0.3943292796611786, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9415397047996521, "reward_repeat_soft_std": 0.05928464233875275, "reward_judge_quality_mean": 0.3462499976158142, "reward_judge_quality_std": 0.10336308926343918, "reward_total_composite_mean": 0.6456738710403442, "reward_total_composite_std": 0.1671849638223648} {"timestamp_utc": "2026-04-13T05:31:10Z", "mode": "train", "global_step": 2856, "epoch": 0.2868910095429432, "loss": -0.0345, "grad_norm": 15.638703346252441, "learning_rate": 1.3484848484848486e-06, "num_tokens": 5329561.0, "completions/mean_length": 22.375, "completions/min_length": 20.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.375, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.9614608883857727, "rewards/meter/std": 0.01524702925235033, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.803032398223877, "rewards/total_composite/std": 0.0177935678511858, "reward": 0.803032398223877, "reward_std": 0.017793571576476097, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07949800044298172, "sampling/sampling_logp_difference/max": 1.929366111755371, "sampling/importance_sampling_ratio/min": 0.14524023234844208, "sampling/importance_sampling_ratio/mean": 1.004106044769287, "sampling/importance_sampling_ratio/max": 1.7302088737487793, "entropy": 0.3131856005638838, "clip_ratio/low_mean": 0.012500000186264515, "clip_ratio/low_min": 0.012500000186264515, "clip_ratio/high_mean": 0.05996047519147396, "clip_ratio/high_max": 0.05996047519147396, "clip_ratio/region_mean": 0.07246047537773848, "reward_total_mean": 0.803032398223877, "reward_meter_mean": 0.9614608883857727, "reward_meter_std": 0.01524702925235033, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.803032398223877, "reward_total_composite_std": 0.0177935678511858} {"timestamp_utc": "2026-04-13T05:31:18Z", "mode": "train", "global_step": 2857, "epoch": 0.28699146157709693, "loss": 0.0635, "grad_norm": 5.9549880027771, "learning_rate": 1.3454545454545457e-06, "num_tokens": 5332441.0, "completions/mean_length": 167.0, "completions/min_length": 143.0, "completions/max_length": 183.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 167.0, "completions/min_terminated_length": 143.0, "completions/max_terminated_length": 183.0, "rewards/meter/mean": 0.9837148189544678, "rewards/meter/std": 0.011232483200728893, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7219884395599365, "rewards/repeat_soft/std": 0.07730896025896072, "rewards/judge_quality/mean": 0.30124998092651367, "rewards/judge_quality/std": 0.10398317128419876, "rewards/total_composite/mean": 0.7552455067634583, "rewards/total_composite/std": 0.036325663328170776, "reward": 0.7552455067634583, "reward_std": 0.03632565215229988, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06489819288253784, "sampling/sampling_logp_difference/max": 3.304546356201172, "sampling/importance_sampling_ratio/min": 0.03671586140990257, "sampling/importance_sampling_ratio/mean": 1.0061206817626953, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.26499635726213455, "clip_ratio/low_mean": 0.032567653339356184, "clip_ratio/low_min": 0.032567653339356184, "clip_ratio/high_mean": 0.02215423434972763, "clip_ratio/high_max": 0.02215423434972763, "clip_ratio/region_mean": 0.054721887689083815, "reward_total_mean": 0.7552455067634583, "reward_meter_mean": 0.9837148189544678, "reward_meter_std": 0.011232483200728893, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7219884395599365, "reward_repeat_soft_std": 0.07730896025896072, "reward_judge_quality_mean": 0.30124998092651367, "reward_judge_quality_std": 0.10398317128419876, "reward_total_composite_mean": 0.7552455067634583, "reward_total_composite_std": 0.036325663328170776} {"timestamp_utc": "2026-04-13T05:31:25Z", "mode": "train", "global_step": 2858, "epoch": 0.28709191361125064, "loss": 0.0021, "grad_norm": 4.557533264160156, "learning_rate": 1.3424242424242425e-06, "num_tokens": 5335207.0, "completions/mean_length": 142.75, "completions/min_length": 127.0, "completions/max_length": 156.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.75, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 156.0, "rewards/meter/mean": 0.9804261922836304, "rewards/meter/std": 0.034320197999477386, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.893726646900177, "rewards/repeat_soft/std": 0.048868294805288315, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.825314462184906, "rewards/total_composite/std": 0.05656137317419052, "reward": 0.825314462184906, "reward_std": 0.05656137689948082, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06753486394882202, "sampling/sampling_logp_difference/max": 1.5885062217712402, "sampling/importance_sampling_ratio/min": 0.22704309225082397, "sampling/importance_sampling_ratio/mean": 1.0057259798049927, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31900015845894814, "clip_ratio/low_mean": 0.04102289071306586, "clip_ratio/low_min": 0.04102289071306586, "clip_ratio/high_mean": 0.007867133244872093, "clip_ratio/high_max": 0.007867133244872093, "clip_ratio/region_mean": 0.048890023957937956, "reward_total_mean": 0.825314462184906, "reward_meter_mean": 0.9804261922836304, "reward_meter_std": 0.034320197999477386, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.893726646900177, "reward_repeat_soft_std": 0.048868294805288315, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.825314462184906, "reward_total_composite_std": 0.05656137317419052} {"timestamp_utc": "2026-04-13T05:31:33Z", "mode": "train", "global_step": 2859, "epoch": 0.28719236564540435, "loss": 0.0458, "grad_norm": 8.043806076049805, "learning_rate": 1.3393939393939395e-06, "num_tokens": 5337862.0, "completions/mean_length": 132.875, "completions/min_length": 121.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.875, "completions/min_terminated_length": 121.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.9921788573265076, "rewards/meter/std": 0.0030395847279578447, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9545479416847229, "rewards/repeat_soft/std": 0.01746583916246891, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.8228102922439575, "rewards/total_composite/std": 0.03912763297557831, "reward": 0.8228102922439575, "reward_std": 0.03912763297557831, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09607156366109848, "sampling/sampling_logp_difference/max": 3.9792590141296387, "sampling/importance_sampling_ratio/min": 0.018699489533901215, "sampling/importance_sampling_ratio/mean": 1.0095312595367432, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4500740095973015, "clip_ratio/low_mean": 0.06599978683516383, "clip_ratio/low_min": 0.06599978683516383, "clip_ratio/high_mean": 0.013429751619696617, "clip_ratio/high_max": 0.013429751619696617, "clip_ratio/region_mean": 0.07942953845486045, "reward_total_mean": 0.8228102922439575, "reward_meter_mean": 0.9921788573265076, "reward_meter_std": 0.0030395847279578447, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9545479416847229, "reward_repeat_soft_std": 0.01746583916246891, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.8228102922439575, "reward_total_composite_std": 0.03912763297557831} {"timestamp_utc": "2026-04-13T05:31:38Z", "mode": "train", "global_step": 2860, "epoch": 0.287292817679558, "loss": 0.0219, "grad_norm": 15.892383575439453, "learning_rate": 1.3363636363636364e-06, "num_tokens": 5339403.0, "completions/mean_length": 26.625, "completions/min_length": 23.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.625, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9843926429748535, "rewards/meter/std": 0.003140920540317893, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9469348192214966, "rewards/repeat_soft/std": 0.02117719128727913, "rewards/judge_quality/mean": 0.42124998569488525, "rewards/judge_quality/std": 0.06998724490404129, "rewards/total_composite/mean": 0.8140451908111572, "rewards/total_composite/std": 0.021197570487856865, "reward": 0.8140451908111572, "reward_std": 0.02119756117463112, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.092268206179142, "sampling/sampling_logp_difference/max": 1.1822967529296875, "sampling/importance_sampling_ratio/min": 0.3065738081932068, "sampling/importance_sampling_ratio/mean": 1.0026642084121704, "sampling/importance_sampling_ratio/max": 1.8334177732467651, "entropy": 0.3792108166962862, "clip_ratio/low_mean": 0.018200549762696028, "clip_ratio/low_min": 0.018200549762696028, "clip_ratio/high_mean": 0.0793616296723485, "clip_ratio/high_max": 0.0793616296723485, "clip_ratio/region_mean": 0.09756217943504453, "reward_total_mean": 0.8140451908111572, "reward_meter_mean": 0.9843926429748535, "reward_meter_std": 0.003140920540317893, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9469348192214966, "reward_repeat_soft_std": 0.02117719128727913, "reward_judge_quality_mean": 0.42124998569488525, "reward_judge_quality_std": 0.06998724490404129, "reward_total_composite_mean": 0.8140451908111572, "reward_total_composite_std": 0.021197570487856865} {"timestamp_utc": "2026-04-13T05:31:45Z", "mode": "train", "global_step": 2861, "epoch": 0.2873932697137117, "loss": 0.0296, "grad_norm": 11.185647964477539, "learning_rate": 1.3333333333333334e-06, "num_tokens": 5340947.0, "completions/mean_length": 36.0, "completions/min_length": 31.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.9869133234024048, "rewards/meter/std": 0.006815631408244371, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9603985548019409, "rewards/repeat_soft/std": 0.004229877609759569, "rewards/judge_quality/mean": 0.36375001072883606, "rewards/judge_quality/std": 0.13265825808048248, "rewards/total_composite/mean": 0.799275815486908, "rewards/total_composite/std": 0.03923248127102852, "reward": 0.799275815486908, "reward_std": 0.039232492446899414, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09196975827217102, "sampling/sampling_logp_difference/max": 1.021152377128601, "sampling/importance_sampling_ratio/min": 0.3601796329021454, "sampling/importance_sampling_ratio/mean": 1.0231391191482544, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5177741535007954, "clip_ratio/low_mean": 0.013523391913622618, "clip_ratio/low_min": 0.013523391913622618, "clip_ratio/high_mean": 0.05563390371389687, "clip_ratio/high_max": 0.05563390371389687, "clip_ratio/region_mean": 0.06915729562751949, "reward_total_mean": 0.799275815486908, "reward_meter_mean": 0.9869133234024048, "reward_meter_std": 0.006815631408244371, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9603985548019409, "reward_repeat_soft_std": 0.004229877609759569, "reward_judge_quality_mean": 0.36375001072883606, "reward_judge_quality_std": 0.13265825808048248, "reward_total_composite_mean": 0.799275815486908, "reward_total_composite_std": 0.03923248127102852} {"timestamp_utc": "2026-04-13T05:31:51Z", "mode": "train", "global_step": 2862, "epoch": 0.2874937217478654, "loss": 0.0219, "grad_norm": 9.704285621643066, "learning_rate": 1.3303030303030305e-06, "num_tokens": 5342763.0, "completions/mean_length": 64.0, "completions/min_length": 58.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9819257259368896, "rewards/meter/std": 0.012724997475743294, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9840766191482544, "rewards/repeat_soft/std": 0.009214661084115505, "rewards/judge_quality/mean": 0.7737500667572021, "rewards/judge_quality/std": 0.2260807603597641, "rewards/total_composite/mean": 0.9223992824554443, "rewards/total_composite/std": 0.0646890178322792, "reward": 0.9223992824554443, "reward_std": 0.06468900293111801, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08189209550619125, "sampling/sampling_logp_difference/max": 1.2569150924682617, "sampling/importance_sampling_ratio/min": 0.2845304310321808, "sampling/importance_sampling_ratio/mean": 1.0055880546569824, "sampling/importance_sampling_ratio/max": 1.8449857234954834, "entropy": 0.4223451055586338, "clip_ratio/low_mean": 0.028206221293658018, "clip_ratio/low_min": 0.028206221293658018, "clip_ratio/high_mean": 0.056766049237921834, "clip_ratio/high_max": 0.056766049237921834, "clip_ratio/region_mean": 0.08497227053157985, "reward_total_mean": 0.9223992824554443, "reward_meter_mean": 0.9819257259368896, "reward_meter_std": 0.012724997475743294, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9840766191482544, "reward_repeat_soft_std": 0.009214661084115505, "reward_judge_quality_mean": 0.7737500667572021, "reward_judge_quality_std": 0.2260807603597641, "reward_total_composite_mean": 0.9223992824554443, "reward_total_composite_std": 0.0646890178322792} {"timestamp_utc": "2026-04-13T05:31:58Z", "mode": "train", "global_step": 2863, "epoch": 0.28759417378201907, "loss": -0.0033, "grad_norm": 6.860640525817871, "learning_rate": 1.3272727272727273e-06, "num_tokens": 5345081.0, "completions/mean_length": 101.75, "completions/min_length": 92.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.75, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.9921649694442749, "rewards/meter/std": 0.0031941698398441076, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7023830413818359, "rewards/repeat_soft/std": 0.056167151778936386, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.8114625215530396, "rewards/total_composite/std": 0.05212903395295143, "reward": 0.8114625215530396, "reward_std": 0.05212903767824173, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06863349676132202, "sampling/sampling_logp_difference/max": 1.34751296043396, "sampling/importance_sampling_ratio/min": 0.2598858177661896, "sampling/importance_sampling_ratio/mean": 1.0074833631515503, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.34010471031069756, "clip_ratio/low_mean": 0.0596855313051492, "clip_ratio/low_min": 0.0596855313051492, "clip_ratio/high_mean": 0.010514019057154655, "clip_ratio/high_max": 0.010514019057154655, "clip_ratio/region_mean": 0.07019955036230385, "reward_total_mean": 0.8114625215530396, "reward_meter_mean": 0.9921649694442749, "reward_meter_std": 0.0031941698398441076, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7023830413818359, "reward_repeat_soft_std": 0.056167151778936386, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.8114625215530396, "reward_total_composite_std": 0.05212903395295143} {"timestamp_utc": "2026-04-13T05:32:05Z", "mode": "train", "global_step": 2864, "epoch": 0.2876946258161728, "loss": 0.0158, "grad_norm": 8.16102409362793, "learning_rate": 1.3242424242424243e-06, "num_tokens": 5346879.0, "completions/mean_length": 60.75, "completions/min_length": 57.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.75, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9913870096206665, "rewards/meter/std": 0.0012875342508777976, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9691686630249023, "rewards/repeat_soft/std": 0.039196472615003586, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8246660232543945, "rewards/total_composite/std": 0.004701596684753895, "reward": 0.8246660232543945, "reward_std": 0.004701598081737757, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07313408702611923, "sampling/sampling_logp_difference/max": 2.2205135822296143, "sampling/importance_sampling_ratio/min": 0.10855334252119064, "sampling/importance_sampling_ratio/mean": 1.0174144506454468, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.373876940459013, "clip_ratio/low_mean": 0.03559185424819589, "clip_ratio/low_min": 0.03559185424819589, "clip_ratio/high_mean": 0.03247844334691763, "clip_ratio/high_max": 0.03247844334691763, "clip_ratio/region_mean": 0.06807029759511352, "reward_total_mean": 0.8246660232543945, "reward_meter_mean": 0.9913870096206665, "reward_meter_std": 0.0012875342508777976, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9691686630249023, "reward_repeat_soft_std": 0.039196472615003586, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8246660232543945, "reward_total_composite_std": 0.004701596684753895} {"timestamp_utc": "2026-04-13T05:32:11Z", "mode": "train", "global_step": 2865, "epoch": 0.2877950778503265, "loss": 0.0462, "grad_norm": 13.091713905334473, "learning_rate": 1.3212121212121212e-06, "num_tokens": 5348496.0, "completions/mean_length": 42.125, "completions/min_length": 39.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.125, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.9546480178833008, "rewards/meter/std": 0.021640488877892494, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9469744563102722, "rewards/repeat_soft/std": 0.08367584645748138, "rewards/judge_quality/mean": 0.5737500190734863, "rewards/judge_quality/std": 0.3009004592895508, "rewards/total_composite/mean": 0.8464140892028809, "rewards/total_composite/std": 0.09949629008769989, "reward": 0.8464140892028809, "reward_std": 0.0994962826371193, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07842469960451126, "sampling/sampling_logp_difference/max": 1.737884283065796, "sampling/importance_sampling_ratio/min": 0.17589214444160461, "sampling/importance_sampling_ratio/mean": 0.9948875308036804, "sampling/importance_sampling_ratio/max": 1.7502397298812866, "entropy": 0.3780256174504757, "clip_ratio/low_mean": 0.0322855613194406, "clip_ratio/low_min": 0.0322855613194406, "clip_ratio/high_mean": 0.030958625953644514, "clip_ratio/high_max": 0.030958625953644514, "clip_ratio/region_mean": 0.06324418727308512, "reward_total_mean": 0.8464140892028809, "reward_meter_mean": 0.9546480178833008, "reward_meter_std": 0.021640488877892494, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9469744563102722, "reward_repeat_soft_std": 0.08367584645748138, "reward_judge_quality_mean": 0.5737500190734863, "reward_judge_quality_std": 0.3009004592895508, "reward_total_composite_mean": 0.8464140892028809, "reward_total_composite_std": 0.09949629008769989} {"timestamp_utc": "2026-04-13T05:32:18Z", "mode": "train", "global_step": 2866, "epoch": 0.28789552988448014, "loss": 0.0027, "grad_norm": 8.34969711303711, "learning_rate": 1.3181818181818182e-06, "num_tokens": 5350934.0, "completions/mean_length": 128.75, "completions/min_length": 113.0, "completions/max_length": 139.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.75, "completions/min_terminated_length": 113.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.700084924697876, "rewards/meter/std": 0.40735405683517456, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9316814541816711, "rewards/repeat_soft/std": 0.03759566694498062, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.7029563784599304, "rewards/total_composite/std": 0.15571638941764832, "reward": 0.7029563784599304, "reward_std": 0.15571637451648712, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09754397720098495, "sampling/sampling_logp_difference/max": 3.6338424682617188, "sampling/importance_sampling_ratio/min": 0.02641449309885502, "sampling/importance_sampling_ratio/mean": 1.004346489906311, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5075479298830032, "clip_ratio/low_mean": 0.028251097071915865, "clip_ratio/low_min": 0.028251097071915865, "clip_ratio/high_mean": 0.06177929975092411, "clip_ratio/high_max": 0.06177929975092411, "clip_ratio/region_mean": 0.09003039682283998, "reward_total_mean": 0.7029563784599304, "reward_meter_mean": 0.700084924697876, "reward_meter_std": 0.40735405683517456, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9316814541816711, "reward_repeat_soft_std": 0.03759566694498062, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.7029563784599304, "reward_total_composite_std": 0.15571638941764832} {"timestamp_utc": "2026-04-13T05:32:30Z", "mode": "train", "global_step": 2867, "epoch": 0.28799598191863385, "loss": -0.0915, "grad_norm": 1.2201337814331055, "learning_rate": 1.315151515151515e-06, "num_tokens": 5352238.0, "completions/mean_length": 89.0, "completions/min_length": 27.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 28.571430206298828, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.87468421459198, "rewards/meter/std": 0.31450024247169495, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9296945333480835, "rewards/repeat_soft/std": 0.0453413762152195, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.2876474857330322, "rewards/total_composite/mean": 0.7503699064254761, "rewards/total_composite/std": 0.3113168776035309, "reward": 0.7503699064254761, "reward_std": 0.3113168478012085, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0907573401927948, "sampling/sampling_logp_difference/max": 1.2196414470672607, "sampling/importance_sampling_ratio/min": 0.2953360378742218, "sampling/importance_sampling_ratio/mean": 0.9899657964706421, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.33191921561956406, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06660053040832281, "clip_ratio/high_max": 0.06660053040832281, "clip_ratio/region_mean": 0.06660053040832281, "reward_total_mean": 0.7503699064254761, "reward_meter_mean": 0.87468421459198, "reward_meter_std": 0.31450024247169495, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9296945333480835, "reward_repeat_soft_std": 0.0453413762152195, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.2876474857330322, "reward_total_composite_mean": 0.7503699064254761, "reward_total_composite_std": 0.3113168776035309} {"timestamp_utc": "2026-04-13T05:32:36Z", "mode": "train", "global_step": 2868, "epoch": 0.28809643395278756, "loss": 0.0229, "grad_norm": 14.365790367126465, "learning_rate": 1.3121212121212123e-06, "num_tokens": 5353593.0, "completions/mean_length": 35.375, "completions/min_length": 32.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.375, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.9851468801498413, "rewards/meter/std": 0.03197707608342171, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9456367492675781, "rewards/repeat_soft/std": 0.027710048481822014, "rewards/judge_quality/mean": 0.7437499761581421, "rewards/judge_quality/std": 0.2432481348514557, "rewards/total_composite/mean": 0.9110047817230225, "rewards/total_composite/std": 0.07090591639280319, "reward": 0.9110047817230225, "reward_std": 0.0709058865904808, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13062264025211334, "sampling/sampling_logp_difference/max": 1.6573337316513062, "sampling/importance_sampling_ratio/min": 0.19064661860466003, "sampling/importance_sampling_ratio/mean": 1.0023672580718994, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6341809183359146, "clip_ratio/low_mean": 0.033618551678955555, "clip_ratio/low_min": 0.033618551678955555, "clip_ratio/high_mean": 0.08295401744544506, "clip_ratio/high_max": 0.08295401744544506, "clip_ratio/region_mean": 0.11657256912440062, "reward_total_mean": 0.9110047817230225, "reward_meter_mean": 0.9851468801498413, "reward_meter_std": 0.03197707608342171, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9456367492675781, "reward_repeat_soft_std": 0.027710048481822014, "reward_judge_quality_mean": 0.7437499761581421, "reward_judge_quality_std": 0.2432481348514557, "reward_total_composite_mean": 0.9110047817230225, "reward_total_composite_std": 0.07090591639280319} {"timestamp_utc": "2026-04-13T05:32:47Z", "mode": "train", "global_step": 2869, "epoch": 0.2881968859869412, "loss": -0.1346, "grad_norm": 1.6698946952819824, "learning_rate": 1.3090909090909093e-06, "num_tokens": 5355317.0, "completions/mean_length": 106.5, "completions/min_length": 44.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 48.57143020629883, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.7643810510635376, "rewards/meter/std": 0.3075852394104004, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9731877446174622, "rewards/repeat_soft/std": 0.05019199475646019, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.25560852885246277, "rewards/total_composite/mean": 0.6762582063674927, "rewards/total_composite/std": 0.2755482792854309, "reward": 0.6762582063674927, "reward_std": 0.2755482792854309, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10221295058727264, "sampling/sampling_logp_difference/max": 1.1670911312103271, "sampling/importance_sampling_ratio/min": 0.31465062499046326, "sampling/importance_sampling_ratio/mean": 0.9862748384475708, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3559533469378948, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.1070110946893692, "clip_ratio/high_max": 0.1070110946893692, "clip_ratio/region_mean": 0.1070110946893692, "reward_total_mean": 0.6762582063674927, "reward_meter_mean": 0.7643810510635376, "reward_meter_std": 0.3075852394104004, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9731877446174622, "reward_repeat_soft_std": 0.05019199475646019, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.25560852885246277, "reward_total_composite_mean": 0.6762582063674927, "reward_total_composite_std": 0.2755482792854309} {"timestamp_utc": "2026-04-13T05:32:53Z", "mode": "train", "global_step": 2870, "epoch": 0.2882973380210949, "loss": 0.0007, "grad_norm": 6.92059850692749, "learning_rate": 1.3060606060606062e-06, "num_tokens": 5357136.0, "completions/mean_length": 62.375, "completions/min_length": 57.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.375, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9162801504135132, "rewards/meter/std": 0.21598497033119202, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9798964262008667, "rewards/repeat_soft/std": 0.013196216896176338, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.8061907291412354, "rewards/total_composite/std": 0.11701639741659164, "reward": 0.8061907291412354, "reward_std": 0.11701638996601105, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09638921916484833, "sampling/sampling_logp_difference/max": 1.8391332626342773, "sampling/importance_sampling_ratio/min": 0.15895512700080872, "sampling/importance_sampling_ratio/mean": 0.9981589317321777, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45658958330750465, "clip_ratio/low_mean": 0.012096773833036423, "clip_ratio/low_min": 0.012096773833036423, "clip_ratio/high_mean": 0.10231364332139492, "clip_ratio/high_max": 0.10231364332139492, "clip_ratio/region_mean": 0.11441041715443134, "reward_total_mean": 0.8061907291412354, "reward_meter_mean": 0.9162801504135132, "reward_meter_std": 0.21598497033119202, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9798964262008667, "reward_repeat_soft_std": 0.013196216896176338, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.8061907291412354, "reward_total_composite_std": 0.11701639741659164} {"timestamp_utc": "2026-04-13T05:33:00Z", "mode": "train", "global_step": 2871, "epoch": 0.2883977900552486, "loss": -0.0157, "grad_norm": 7.710918426513672, "learning_rate": 1.3030303030303032e-06, "num_tokens": 5359551.0, "completions/mean_length": 113.875, "completions/min_length": 97.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.875, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.9929041266441345, "rewards/meter/std": 0.003174637211486697, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8966676592826843, "rewards/repeat_soft/std": 0.03810112923383713, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.7895986437797546, "rewards/total_composite/std": 0.03458696976304054, "reward": 0.7895986437797546, "reward_std": 0.03458697348833084, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08139578998088837, "sampling/sampling_logp_difference/max": 1.3726763725280762, "sampling/importance_sampling_ratio/min": 0.25342777371406555, "sampling/importance_sampling_ratio/mean": 1.0117920637130737, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37618500739336014, "clip_ratio/low_mean": 0.021166696213185787, "clip_ratio/low_min": 0.021166696213185787, "clip_ratio/high_mean": 0.058052017353475094, "clip_ratio/high_max": 0.058052017353475094, "clip_ratio/region_mean": 0.07921871356666088, "reward_total_mean": 0.7895986437797546, "reward_meter_mean": 0.9929041266441345, "reward_meter_std": 0.003174637211486697, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8966676592826843, "reward_repeat_soft_std": 0.03810112923383713, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.7895986437797546, "reward_total_composite_std": 0.03458696976304054} {"timestamp_utc": "2026-04-13T05:33:07Z", "mode": "train", "global_step": 2872, "epoch": 0.28849824208940233, "loss": 0.0133, "grad_norm": 8.787920951843262, "learning_rate": 1.3e-06, "num_tokens": 5361159.0, "completions/mean_length": 48.0, "completions/min_length": 44.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.0, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.8083385825157166, "rewards/meter/std": 0.2982751727104187, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9478122591972351, "rewards/repeat_soft/std": 0.04613257199525833, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7367836236953735, "rewards/total_composite/std": 0.13805216550827026, "reward": 0.7367836236953735, "reward_std": 0.13805215060710907, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0682084783911705, "sampling/sampling_logp_difference/max": 1.355751395225525, "sampling/importance_sampling_ratio/min": 0.28668567538261414, "sampling/importance_sampling_ratio/mean": 1.0110427141189575, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3187622092664242, "clip_ratio/low_mean": 0.017701049335300922, "clip_ratio/low_min": 0.017701049335300922, "clip_ratio/high_mean": 0.036478279856964946, "clip_ratio/high_max": 0.036478279856964946, "clip_ratio/region_mean": 0.05417932919226587, "reward_total_mean": 0.7367836236953735, "reward_meter_mean": 0.8083385825157166, "reward_meter_std": 0.2982751727104187, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9478122591972351, "reward_repeat_soft_std": 0.04613257199525833, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7367836236953735, "reward_total_composite_std": 0.13805216550827026} {"timestamp_utc": "2026-04-13T05:33:14Z", "mode": "train", "global_step": 2873, "epoch": 0.288598694123556, "loss": 0.0315, "grad_norm": 10.918354988098145, "learning_rate": 1.296969696969697e-06, "num_tokens": 5363075.0, "completions/mean_length": 59.5, "completions/min_length": 56.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.5, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9825429916381836, "rewards/meter/std": 0.013457007706165314, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9806480407714844, "rewards/repeat_soft/std": 0.013274616561830044, "rewards/judge_quality/mean": 0.41749998927116394, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.8154591917991638, "rewards/total_composite/std": 0.019755462184548378, "reward": 0.8154591917991638, "reward_std": 0.019755471497774124, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06735122948884964, "sampling/sampling_logp_difference/max": 1.1581006050109863, "sampling/importance_sampling_ratio/min": 0.33755815029144287, "sampling/importance_sampling_ratio/mean": 1.0108702182769775, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.30182773247361183, "clip_ratio/low_mean": 0.012404318898916245, "clip_ratio/low_min": 0.012404318898916245, "clip_ratio/high_mean": 0.050770625937730074, "clip_ratio/high_max": 0.050770625937730074, "clip_ratio/region_mean": 0.06317494483664632, "reward_total_mean": 0.8154591917991638, "reward_meter_mean": 0.9825429916381836, "reward_meter_std": 0.013457007706165314, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9806480407714844, "reward_repeat_soft_std": 0.013274616561830044, "reward_judge_quality_mean": 0.41749998927116394, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.8154591917991638, "reward_total_composite_std": 0.019755462184548378} {"timestamp_utc": "2026-04-13T05:33:20Z", "mode": "train", "global_step": 2874, "epoch": 0.2886991461577097, "loss": 0.0638, "grad_norm": 10.287940979003906, "learning_rate": 1.2939393939393941e-06, "num_tokens": 5364734.0, "completions/mean_length": 54.375, "completions/min_length": 44.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.375, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9588642120361328, "rewards/meter/std": 0.07219544798135757, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.923143744468689, "rewards/repeat_soft/std": 0.036814793944358826, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.8748033046722412, "rewards/total_composite/std": 0.07677802443504333, "reward": 0.8748033046722412, "reward_std": 0.07677802443504333, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08703973144292831, "sampling/sampling_logp_difference/max": 1.1947004795074463, "sampling/importance_sampling_ratio/min": 0.3027946352958679, "sampling/importance_sampling_ratio/mean": 1.0132509469985962, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3526206947863102, "clip_ratio/low_mean": 0.036848253570497036, "clip_ratio/low_min": 0.036848253570497036, "clip_ratio/high_mean": 0.04196529369801283, "clip_ratio/high_max": 0.04196529369801283, "clip_ratio/region_mean": 0.07881354726850986, "reward_total_mean": 0.8748033046722412, "reward_meter_mean": 0.9588642120361328, "reward_meter_std": 0.07219544798135757, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.923143744468689, "reward_repeat_soft_std": 0.036814793944358826, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.8748033046722412, "reward_total_composite_std": 0.07677802443504333} {"timestamp_utc": "2026-04-13T05:33:28Z", "mode": "train", "global_step": 2875, "epoch": 0.2887995981918634, "loss": -0.0284, "grad_norm": 4.3020806312561035, "learning_rate": 1.290909090909091e-06, "num_tokens": 5367453.0, "completions/mean_length": 145.875, "completions/min_length": 140.0, "completions/max_length": 162.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 145.875, "completions/min_terminated_length": 140.0, "completions/max_terminated_length": 162.0, "rewards/meter/mean": 0.9896005392074585, "rewards/meter/std": 0.00418513547629118, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.616639256477356, "rewards/repeat_soft/std": 0.038179948925971985, "rewards/judge_quality/mean": 0.45249998569488525, "rewards/judge_quality/std": 0.21224987506866455, "rewards/total_composite/mean": 0.7927342057228088, "rewards/total_composite/std": 0.06566966325044632, "reward": 0.7927342057228088, "reward_std": 0.06566966325044632, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.055026985704898834, "sampling/sampling_logp_difference/max": 1.359029769897461, "sampling/importance_sampling_ratio/min": 0.25690990686416626, "sampling/importance_sampling_ratio/mean": 1.0079708099365234, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.26621589437127113, "clip_ratio/low_mean": 0.04363152221776545, "clip_ratio/low_min": 0.04363152221776545, "clip_ratio/high_mean": 0.005401234608143568, "clip_ratio/high_max": 0.005401234608143568, "clip_ratio/region_mean": 0.04903275682590902, "reward_total_mean": 0.7927342057228088, "reward_meter_mean": 0.9896005392074585, "reward_meter_std": 0.00418513547629118, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.616639256477356, "reward_repeat_soft_std": 0.038179948925971985, "reward_judge_quality_mean": 0.45249998569488525, "reward_judge_quality_std": 0.21224987506866455, "reward_total_composite_mean": 0.7927342057228088, "reward_total_composite_std": 0.06566966325044632} {"timestamp_utc": "2026-04-13T05:33:35Z", "mode": "train", "global_step": 2876, "epoch": 0.28890005022601706, "loss": 0.059, "grad_norm": 10.536057472229004, "learning_rate": 1.287878787878788e-06, "num_tokens": 5369203.0, "completions/mean_length": 61.75, "completions/min_length": 55.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.75, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.8578072786331177, "rewards/meter/std": 0.2478659301996231, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9809980392456055, "rewards/repeat_soft/std": 0.017254840582609177, "rewards/judge_quality/mean": 0.6200000047683716, "rewards/judge_quality/std": 0.22677870094776154, "rewards/total_composite/mean": 0.8201130628585815, "rewards/total_composite/std": 0.1101069301366806, "reward": 0.8201130628585815, "reward_std": 0.1101069301366806, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09682635962963104, "sampling/sampling_logp_difference/max": 1.2903313636779785, "sampling/importance_sampling_ratio/min": 0.27517956495285034, "sampling/importance_sampling_ratio/mean": 1.0041035413742065, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47193365544080734, "clip_ratio/low_mean": 0.038860134314745665, "clip_ratio/low_min": 0.038860134314745665, "clip_ratio/high_mean": 0.030492424499243498, "clip_ratio/high_max": 0.030492424499243498, "clip_ratio/region_mean": 0.06935255881398916, "reward_total_mean": 0.8201130628585815, "reward_meter_mean": 0.8578072786331177, "reward_meter_std": 0.2478659301996231, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9809980392456055, "reward_repeat_soft_std": 0.017254840582609177, "reward_judge_quality_mean": 0.6200000047683716, "reward_judge_quality_std": 0.22677870094776154, "reward_total_composite_mean": 0.8201130628585815, "reward_total_composite_std": 0.1101069301366806} {"timestamp_utc": "2026-04-13T05:33:43Z", "mode": "train", "global_step": 2877, "epoch": 0.28900050226017077, "loss": -0.0023, "grad_norm": 4.734841823577881, "learning_rate": 1.2848484848484848e-06, "num_tokens": 5372638.0, "completions/mean_length": 209.375, "completions/min_length": 197.0, "completions/max_length": 221.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 209.375, "completions/min_terminated_length": 197.0, "completions/max_terminated_length": 221.0, "rewards/meter/mean": 0.9946702718734741, "rewards/meter/std": 0.001620069844648242, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9271340370178223, "rewards/repeat_soft/std": 0.035427603870630264, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.8213150501251221, "rewards/total_composite/std": 0.035046450793743134, "reward": 0.8213150501251221, "reward_std": 0.035046447068452835, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08987537771463394, "sampling/sampling_logp_difference/max": 2.339369773864746, "sampling/importance_sampling_ratio/min": 0.09638836234807968, "sampling/importance_sampling_ratio/mean": 0.9953790903091431, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36706142500042915, "clip_ratio/low_mean": 0.06457775738090277, "clip_ratio/low_min": 0.06457775738090277, "clip_ratio/high_mean": 0.010613207705318928, "clip_ratio/high_max": 0.010613207705318928, "clip_ratio/region_mean": 0.0751909650862217, "reward_total_mean": 0.8213150501251221, "reward_meter_mean": 0.9946702718734741, "reward_meter_std": 0.001620069844648242, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9271340370178223, "reward_repeat_soft_std": 0.035427603870630264, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.8213150501251221, "reward_total_composite_std": 0.035046450793743134} {"timestamp_utc": "2026-04-13T05:33:50Z", "mode": "train", "global_step": 2878, "epoch": 0.2891009542943245, "loss": 0.0068, "grad_norm": 8.76352596282959, "learning_rate": 1.2818181818181819e-06, "num_tokens": 5374447.0, "completions/mean_length": 63.125, "completions/min_length": 51.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.125, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9837009906768799, "rewards/meter/std": 0.008695956319570541, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.970335841178894, "rewards/repeat_soft/std": 0.0343095064163208, "rewards/judge_quality/mean": 0.6187499761581421, "rewards/judge_quality/std": 0.24976776540279388, "rewards/total_composite/mean": 0.8753240704536438, "rewards/total_composite/std": 0.07291785627603531, "reward": 0.8753240704536438, "reward_std": 0.07291785627603531, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09312830865383148, "sampling/sampling_logp_difference/max": 1.0520033836364746, "sampling/importance_sampling_ratio/min": 0.3492373824119568, "sampling/importance_sampling_ratio/mean": 1.0151546001434326, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4184941127896309, "clip_ratio/low_mean": 0.06309039983898401, "clip_ratio/low_min": 0.06309039983898401, "clip_ratio/high_mean": 0.01929204585030675, "clip_ratio/high_max": 0.01929204585030675, "clip_ratio/region_mean": 0.08238244568929076, "reward_total_mean": 0.8753240704536438, "reward_meter_mean": 0.9837009906768799, "reward_meter_std": 0.008695956319570541, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.970335841178894, "reward_repeat_soft_std": 0.0343095064163208, "reward_judge_quality_mean": 0.6187499761581421, "reward_judge_quality_std": 0.24976776540279388, "reward_total_composite_mean": 0.8753240704536438, "reward_total_composite_std": 0.07291785627603531} {"timestamp_utc": "2026-04-13T05:33:57Z", "mode": "train", "global_step": 2879, "epoch": 0.28920140632847813, "loss": 0.045, "grad_norm": 17.97874641418457, "learning_rate": 1.2787878787878787e-06, "num_tokens": 5376103.0, "completions/mean_length": 32.0, "completions/min_length": 29.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.782668948173523, "rewards/meter/std": 0.349631667137146, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9487870931625366, "rewards/repeat_soft/std": 0.03148152306675911, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.08750510215759277, "rewards/total_composite/mean": 0.7215797305107117, "rewards/total_composite/std": 0.1547526866197586, "reward": 0.7215797305107117, "reward_std": 0.1547527015209198, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10472270846366882, "sampling/sampling_logp_difference/max": 1.4045811891555786, "sampling/importance_sampling_ratio/min": 0.2454698383808136, "sampling/importance_sampling_ratio/mean": 1.0104349851608276, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5793041810393333, "clip_ratio/low_mean": 0.02346041053533554, "clip_ratio/low_min": 0.02346041053533554, "clip_ratio/high_mean": 0.09351155860349536, "clip_ratio/high_max": 0.09351155860349536, "clip_ratio/region_mean": 0.1169719691388309, "reward_total_mean": 0.7215797305107117, "reward_meter_mean": 0.782668948173523, "reward_meter_std": 0.349631667137146, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9487870931625366, "reward_repeat_soft_std": 0.03148152306675911, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.08750510215759277, "reward_total_composite_mean": 0.7215797305107117, "reward_total_composite_std": 0.1547526866197586} {"timestamp_utc": "2026-04-13T05:34:03Z", "mode": "train", "global_step": 2880, "epoch": 0.28930185836263184, "loss": 0.0187, "grad_norm": 15.108933448791504, "learning_rate": 1.2757575757575758e-06, "num_tokens": 5377718.0, "completions/mean_length": 38.875, "completions/min_length": 33.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.6860255002975464, "rewards/meter/std": 0.33144068717956543, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9777984619140625, "rewards/repeat_soft/std": 0.02261141873896122, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.7574913501739502, "rewards/total_composite/std": 0.09730640798807144, "reward": 0.7574913501739502, "reward_std": 0.09730640053749084, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10315179079771042, "sampling/sampling_logp_difference/max": 2.551356315612793, "sampling/importance_sampling_ratio/min": 0.07797583192586899, "sampling/importance_sampling_ratio/mean": 0.9931850433349609, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.38134334050118923, "clip_ratio/low_mean": 0.027346637099981308, "clip_ratio/low_min": 0.027346637099981308, "clip_ratio/high_mean": 0.05540478276088834, "clip_ratio/high_max": 0.05540478276088834, "clip_ratio/region_mean": 0.08275141986086965, "reward_total_mean": 0.7574913501739502, "reward_meter_mean": 0.6860255002975464, "reward_meter_std": 0.33144068717956543, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9777984619140625, "reward_repeat_soft_std": 0.02261141873896122, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.7574913501739502, "reward_total_composite_std": 0.09730640798807144} {"timestamp_utc": "2026-04-13T05:34:10Z", "mode": "train", "global_step": 2881, "epoch": 0.28940231039678554, "loss": 0.0466, "grad_norm": 10.253753662109375, "learning_rate": 1.2727272727272728e-06, "num_tokens": 5379423.0, "completions/mean_length": 58.125, "completions/min_length": 55.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.125, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9829159379005432, "rewards/meter/std": 0.006661101244390011, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9279118776321411, "rewards/repeat_soft/std": 0.03284928575158119, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.9236032962799072, "rewards/total_composite/std": 0.067374587059021, "reward": 0.9236032962799072, "reward_std": 0.06737459450960159, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08127416670322418, "sampling/sampling_logp_difference/max": 1.333930492401123, "sampling/importance_sampling_ratio/min": 0.26343977451324463, "sampling/importance_sampling_ratio/mean": 1.0114614963531494, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.41675421223044395, "clip_ratio/low_mean": 0.010209235595539212, "clip_ratio/low_min": 0.010209235595539212, "clip_ratio/high_mean": 0.07315525691956282, "clip_ratio/high_max": 0.07315525691956282, "clip_ratio/region_mean": 0.08336449251510203, "reward_total_mean": 0.9236032962799072, "reward_meter_mean": 0.9829159379005432, "reward_meter_std": 0.006661101244390011, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9279118776321411, "reward_repeat_soft_std": 0.03284928575158119, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.9236032962799072, "reward_total_composite_std": 0.067374587059021} {"timestamp_utc": "2026-04-13T05:34:17Z", "mode": "train", "global_step": 2882, "epoch": 0.28950276243093925, "loss": 0.0086, "grad_norm": 7.048101425170898, "learning_rate": 1.2696969696969698e-06, "num_tokens": 5382046.0, "completions/mean_length": 144.875, "completions/min_length": 138.0, "completions/max_length": 155.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 144.875, "completions/min_terminated_length": 138.0, "completions/max_terminated_length": 155.0, "rewards/meter/mean": 0.9791258573532104, "rewards/meter/std": 0.012529953382909298, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9084811210632324, "rewards/repeat_soft/std": 0.024421745911240578, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.707208514213562, "rewards/total_composite/std": 0.2858041524887085, "reward": 0.707208514213562, "reward_std": 0.2858041226863861, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09501762688159943, "sampling/sampling_logp_difference/max": 1.3836874961853027, "sampling/importance_sampling_ratio/min": 0.250652551651001, "sampling/importance_sampling_ratio/mean": 1.0030560493469238, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5054912827908993, "clip_ratio/low_mean": 0.011443661525845528, "clip_ratio/low_min": 0.011443661525845528, "clip_ratio/high_mean": 0.08098720759153366, "clip_ratio/high_max": 0.08098720759153366, "clip_ratio/region_mean": 0.09243086911737919, "reward_total_mean": 0.707208514213562, "reward_meter_mean": 0.9791258573532104, "reward_meter_std": 0.012529953382909298, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9084811210632324, "reward_repeat_soft_std": 0.024421745911240578, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.707208514213562, "reward_total_composite_std": 0.2858041524887085} {"timestamp_utc": "2026-04-13T05:34:24Z", "mode": "train", "global_step": 2883, "epoch": 0.2896032144650929, "loss": 0.0246, "grad_norm": 3.924034833908081, "learning_rate": 1.2666666666666669e-06, "num_tokens": 5384571.0, "completions/mean_length": 137.625, "completions/min_length": 135.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 137.625, "completions/min_terminated_length": 135.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.9931238889694214, "rewards/meter/std": 0.002385032130405307, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6117213368415833, "rewards/repeat_soft/std": 0.04362506791949272, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.8215779066085815, "rewards/total_composite/std": 0.07301116734743118, "reward": 0.8215779066085815, "reward_std": 0.07301116734743118, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05447772145271301, "sampling/sampling_logp_difference/max": 2.8183979988098145, "sampling/importance_sampling_ratio/min": 0.05970150977373123, "sampling/importance_sampling_ratio/mean": 1.0031882524490356, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1808122992515564, "clip_ratio/low_mean": 0.03444674285128713, "clip_ratio/low_min": 0.03444674285128713, "clip_ratio/high_mean": 0.0101102942135185, "clip_ratio/high_max": 0.0101102942135185, "clip_ratio/region_mean": 0.04455703706480563, "reward_total_mean": 0.8215779066085815, "reward_meter_mean": 0.9931238889694214, "reward_meter_std": 0.002385032130405307, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6117213368415833, "reward_repeat_soft_std": 0.04362506791949272, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.8215779066085815, "reward_total_composite_std": 0.07301116734743118} {"timestamp_utc": "2026-04-13T05:34:30Z", "mode": "train", "global_step": 2884, "epoch": 0.2897036664992466, "loss": -0.0292, "grad_norm": 11.310540199279785, "learning_rate": 1.2636363636363637e-06, "num_tokens": 5386099.0, "completions/mean_length": 34.0, "completions/min_length": 26.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.0, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.9878422021865845, "rewards/meter/std": 0.007114765699952841, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9463866949081421, "rewards/repeat_soft/std": 0.024802163243293762, "rewards/judge_quality/mean": 0.5012500286102295, "rewards/judge_quality/std": 0.16974246501922607, "rewards/total_composite/mean": 0.8395426273345947, "rewards/total_composite/std": 0.05278245359659195, "reward": 0.8395426273345947, "reward_std": 0.05278245732188225, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10188456624746323, "sampling/sampling_logp_difference/max": 1.18896484375, "sampling/importance_sampling_ratio/min": 0.3045363426208496, "sampling/importance_sampling_ratio/mean": 1.0123069286346436, "sampling/importance_sampling_ratio/max": 1.8741191625595093, "entropy": 0.698822557926178, "clip_ratio/low_mean": 0.0751943071372807, "clip_ratio/low_min": 0.0751943071372807, "clip_ratio/high_mean": 0.016025641933083534, "clip_ratio/high_max": 0.016025641933083534, "clip_ratio/region_mean": 0.09121994907036424, "reward_total_mean": 0.8395426273345947, "reward_meter_mean": 0.9878422021865845, "reward_meter_std": 0.007114765699952841, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9463866949081421, "reward_repeat_soft_std": 0.024802163243293762, "reward_judge_quality_mean": 0.5012500286102295, "reward_judge_quality_std": 0.16974246501922607, "reward_total_composite_mean": 0.8395426273345947, "reward_total_composite_std": 0.05278245359659195} {"timestamp_utc": "2026-04-13T05:34:37Z", "mode": "train", "global_step": 2885, "epoch": 0.2898041185334003, "loss": -0.002, "grad_norm": 8.335785865783691, "learning_rate": 1.2606060606060608e-06, "num_tokens": 5387957.0, "completions/mean_length": 65.25, "completions/min_length": 61.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.25, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.763339638710022, "rewards/meter/std": 0.34170451760292053, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9838434457778931, "rewards/repeat_soft/std": 0.011870156042277813, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.10392305999994278, "rewards/total_composite/mean": 0.7313872575759888, "rewards/total_composite/std": 0.1307065635919571, "reward": 0.7313872575759888, "reward_std": 0.1307065635919571, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10002627223730087, "sampling/sampling_logp_difference/max": 2.1206483840942383, "sampling/importance_sampling_ratio/min": 0.11995382606983185, "sampling/importance_sampling_ratio/mean": 1.0034685134887695, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5338606871664524, "clip_ratio/low_mean": 0.027859986061230302, "clip_ratio/low_min": 0.027859986061230302, "clip_ratio/high_mean": 0.0746710691601038, "clip_ratio/high_max": 0.0746710691601038, "clip_ratio/region_mean": 0.1025310552213341, "reward_total_mean": 0.7313872575759888, "reward_meter_mean": 0.763339638710022, "reward_meter_std": 0.34170451760292053, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9838434457778931, "reward_repeat_soft_std": 0.011870156042277813, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.10392305999994278, "reward_total_composite_mean": 0.7313872575759888, "reward_total_composite_std": 0.1307065635919571} {"timestamp_utc": "2026-04-13T05:34:45Z", "mode": "train", "global_step": 2886, "epoch": 0.289904570567554, "loss": -0.0141, "grad_norm": 4.96796989440918, "learning_rate": 1.2575757575757578e-06, "num_tokens": 5390705.0, "completions/mean_length": 152.5, "completions/min_length": 145.0, "completions/max_length": 164.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 152.5, "completions/min_terminated_length": 145.0, "completions/max_terminated_length": 164.0, "rewards/meter/mean": 0.9607986211776733, "rewards/meter/std": 0.0398171991109848, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8717501163482666, "rewards/repeat_soft/std": 0.08939406275749207, "rewards/judge_quality/mean": 0.3187499940395355, "rewards/judge_quality/std": 0.1397382766008377, "rewards/total_composite/mean": 0.7651593685150146, "rewards/total_composite/std": 0.04230838268995285, "reward": 0.7651593685150146, "reward_std": 0.04230836406350136, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.060319237411022186, "sampling/sampling_logp_difference/max": 1.2911086082458496, "sampling/importance_sampling_ratio/min": 0.2749657928943634, "sampling/importance_sampling_ratio/mean": 0.9961291551589966, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3116326853632927, "clip_ratio/low_mean": 0.01949811028316617, "clip_ratio/low_min": 0.01949811028316617, "clip_ratio/high_mean": 0.04086603503674269, "clip_ratio/high_max": 0.04086603503674269, "clip_ratio/region_mean": 0.06036414531990886, "reward_total_mean": 0.7651593685150146, "reward_meter_mean": 0.9607986211776733, "reward_meter_std": 0.0398171991109848, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8717501163482666, "reward_repeat_soft_std": 0.08939406275749207, "reward_judge_quality_mean": 0.3187499940395355, "reward_judge_quality_std": 0.1397382766008377, "reward_total_composite_mean": 0.7651593685150146, "reward_total_composite_std": 0.04230838268995285} {"timestamp_utc": "2026-04-13T05:34:51Z", "mode": "train", "global_step": 2887, "epoch": 0.2900050226017077, "loss": 0.0065, "grad_norm": 14.388111114501953, "learning_rate": 1.2545454545454546e-06, "num_tokens": 5392372.0, "completions/mean_length": 45.375, "completions/min_length": 44.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.375, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9678476452827454, "rewards/meter/std": 0.020165203139185905, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9826905727386475, "rewards/repeat_soft/std": 0.028312141075730324, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.814300537109375, "rewards/total_composite/std": 0.011280544102191925, "reward": 0.814300537109375, "reward_std": 0.011280539445579052, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07502824068069458, "sampling/sampling_logp_difference/max": 1.3659820556640625, "sampling/importance_sampling_ratio/min": 0.25512999296188354, "sampling/importance_sampling_ratio/mean": 1.0019792318344116, "sampling/importance_sampling_ratio/max": 1.7832525968551636, "entropy": 0.3812127448618412, "clip_ratio/low_mean": 0.011363636702299118, "clip_ratio/low_min": 0.011363636702299118, "clip_ratio/high_mean": 0.04906517802737653, "clip_ratio/high_max": 0.04906517802737653, "clip_ratio/region_mean": 0.06042881472967565, "reward_total_mean": 0.814300537109375, "reward_meter_mean": 0.9678476452827454, "reward_meter_std": 0.020165203139185905, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9826905727386475, "reward_repeat_soft_std": 0.028312141075730324, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.814300537109375, "reward_total_composite_std": 0.011280544102191925} {"timestamp_utc": "2026-04-13T05:34:57Z", "mode": "train", "global_step": 2888, "epoch": 0.2901054746358614, "loss": 0.0361, "grad_norm": 7.0652971267700195, "learning_rate": 1.2515151515151517e-06, "num_tokens": 5394061.0, "completions/mean_length": 66.125, "completions/min_length": 62.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.125, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9913562536239624, "rewards/meter/std": 0.0100261764600873, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9070013761520386, "rewards/repeat_soft/std": 0.10040032863616943, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.813935399055481, "rewards/total_composite/std": 0.008422751910984516, "reward": 0.813935399055481, "reward_std": 0.008422743529081345, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07616420090198517, "sampling/sampling_logp_difference/max": 1.5774364471435547, "sampling/importance_sampling_ratio/min": 0.20650379359722137, "sampling/importance_sampling_ratio/mean": 1.00331711769104, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3950047045946121, "clip_ratio/low_mean": 0.02935606148093939, "clip_ratio/low_min": 0.02935606148093939, "clip_ratio/high_mean": 0.05137645173817873, "clip_ratio/high_max": 0.05137645173817873, "clip_ratio/region_mean": 0.08073251321911812, "reward_total_mean": 0.813935399055481, "reward_meter_mean": 0.9913562536239624, "reward_meter_std": 0.0100261764600873, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9070013761520386, "reward_repeat_soft_std": 0.10040032863616943, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.813935399055481, "reward_total_composite_std": 0.008422751910984516} {"timestamp_utc": "2026-04-13T05:35:03Z", "mode": "train", "global_step": 2889, "epoch": 0.29020592667001505, "loss": 0.0391, "grad_norm": 8.51860523223877, "learning_rate": 1.2484848484848485e-06, "num_tokens": 5395707.0, "completions/mean_length": 58.75, "completions/min_length": 54.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.75, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9754988551139832, "rewards/meter/std": 0.02137601748108864, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8991464972496033, "rewards/repeat_soft/std": 0.07055211067199707, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.9173891544342041, "rewards/total_composite/std": 0.0735919326543808, "reward": 0.9173891544342041, "reward_std": 0.0735919326543808, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07691485434770584, "sampling/sampling_logp_difference/max": 1.2420482635498047, "sampling/importance_sampling_ratio/min": 0.28879210352897644, "sampling/importance_sampling_ratio/mean": 1.0118482112884521, "sampling/importance_sampling_ratio/max": 1.921197772026062, "entropy": 0.47648095712065697, "clip_ratio/low_mean": 0.01871237065643072, "clip_ratio/low_min": 0.01871237065643072, "clip_ratio/high_mean": 0.05694537051022053, "clip_ratio/high_max": 0.05694537051022053, "clip_ratio/region_mean": 0.07565774116665125, "reward_total_mean": 0.9173891544342041, "reward_meter_mean": 0.9754988551139832, "reward_meter_std": 0.02137601748108864, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8991464972496033, "reward_repeat_soft_std": 0.07055211067199707, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.9173891544342041, "reward_total_composite_std": 0.0735919326543808} {"timestamp_utc": "2026-04-13T05:35:10Z", "mode": "train", "global_step": 2890, "epoch": 0.29030637870416875, "loss": 0.08, "grad_norm": 14.345166206359863, "learning_rate": 1.2454545454545456e-06, "num_tokens": 5397236.0, "completions/mean_length": 40.125, "completions/min_length": 34.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.125, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.8256281614303589, "rewards/meter/std": 0.1424683928489685, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9958665370941162, "rewards/repeat_soft/std": 0.0049917735159397125, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7527443170547485, "rewards/total_composite/std": 0.06408947706222534, "reward": 0.7527443170547485, "reward_std": 0.06408947706222534, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11833047866821289, "sampling/sampling_logp_difference/max": 1.5275778770446777, "sampling/importance_sampling_ratio/min": 0.21706078946590424, "sampling/importance_sampling_ratio/mean": 0.9902839660644531, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40577540546655655, "clip_ratio/low_mean": 0.04531746078282595, "clip_ratio/low_min": 0.04531746078282595, "clip_ratio/high_mean": 0.05587300378829241, "clip_ratio/high_max": 0.05587300378829241, "clip_ratio/region_mean": 0.10119046457111835, "reward_total_mean": 0.7527443170547485, "reward_meter_mean": 0.8256281614303589, "reward_meter_std": 0.1424683928489685, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9958665370941162, "reward_repeat_soft_std": 0.0049917735159397125, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7527443170547485, "reward_total_composite_std": 0.06408947706222534} {"timestamp_utc": "2026-04-13T05:35:16Z", "mode": "train", "global_step": 2891, "epoch": 0.29040683073832246, "loss": 0.0237, "grad_norm": 14.597519874572754, "learning_rate": 1.2424242424242424e-06, "num_tokens": 5398669.0, "completions/mean_length": 26.125, "completions/min_length": 22.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.125, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.9507560729980469, "rewards/meter/std": 0.05594087764620781, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9487733840942383, "rewards/repeat_soft/std": 0.016160547733306885, "rewards/judge_quality/mean": 0.5600000023841858, "rewards/judge_quality/std": 0.22258226573467255, "rewards/total_composite/mean": 0.8407175540924072, "rewards/total_composite/std": 0.06755910068750381, "reward": 0.8407175540924072, "reward_std": 0.06755910068750381, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08140888810157776, "sampling/sampling_logp_difference/max": 1.0061919689178467, "sampling/importance_sampling_ratio/min": 0.3656085729598999, "sampling/importance_sampling_ratio/mean": 1.0200306177139282, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.46570419520139694, "clip_ratio/low_mean": 0.07525745406746864, "clip_ratio/low_min": 0.07525745406746864, "clip_ratio/high_mean": 0.0223214291036129, "clip_ratio/high_max": 0.0223214291036129, "clip_ratio/region_mean": 0.09757888317108154, "reward_total_mean": 0.8407175540924072, "reward_meter_mean": 0.9507560729980469, "reward_meter_std": 0.05594087764620781, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9487733840942383, "reward_repeat_soft_std": 0.016160547733306885, "reward_judge_quality_mean": 0.5600000023841858, "reward_judge_quality_std": 0.22258226573467255, "reward_total_composite_mean": 0.8407175540924072, "reward_total_composite_std": 0.06755910068750381} {"timestamp_utc": "2026-04-13T05:35:22Z", "mode": "train", "global_step": 2892, "epoch": 0.2905072827724761, "loss": 0.0525, "grad_norm": 7.624941825866699, "learning_rate": 1.2393939393939394e-06, "num_tokens": 5400378.0, "completions/mean_length": 62.625, "completions/min_length": 56.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.625, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9730619788169861, "rewards/meter/std": 0.04621617868542671, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9088084697723389, "rewards/repeat_soft/std": 0.0905958041548729, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.8171337246894836, "rewards/total_composite/std": 0.041915103793144226, "reward": 0.8171337246894836, "reward_std": 0.04191509261727333, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06498148292303085, "sampling/sampling_logp_difference/max": 1.567911148071289, "sampling/importance_sampling_ratio/min": 0.20848020911216736, "sampling/importance_sampling_ratio/mean": 0.9979663491249084, "sampling/importance_sampling_ratio/max": 1.7331401109695435, "entropy": 0.3131634369492531, "clip_ratio/low_mean": 0.05288756242953241, "clip_ratio/low_min": 0.05288756242953241, "clip_ratio/high_mean": 0.010912698926404119, "clip_ratio/high_max": 0.010912698926404119, "clip_ratio/region_mean": 0.06380026135593653, "reward_total_mean": 0.8171337246894836, "reward_meter_mean": 0.9730619788169861, "reward_meter_std": 0.04621617868542671, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9088084697723389, "reward_repeat_soft_std": 0.0905958041548729, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.8171337246894836, "reward_total_composite_std": 0.041915103793144226} {"timestamp_utc": "2026-04-13T05:35:29Z", "mode": "train", "global_step": 2893, "epoch": 0.2906077348066298, "loss": 0.0264, "grad_norm": 6.554185390472412, "learning_rate": 1.2363636363636365e-06, "num_tokens": 5402146.0, "completions/mean_length": 66.0, "completions/min_length": 60.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9923086166381836, "rewards/meter/std": 0.0017002333188429475, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8601362109184265, "rewards/repeat_soft/std": 0.049438152462244034, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8141774535179138, "rewards/total_composite/std": 0.007985465228557587, "reward": 0.8141774535179138, "reward_std": 0.007985467091202736, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04820971190929413, "sampling/sampling_logp_difference/max": 1.918816089630127, "sampling/importance_sampling_ratio/min": 0.14678062498569489, "sampling/importance_sampling_ratio/mean": 1.0096828937530518, "sampling/importance_sampling_ratio/max": 1.8433479070663452, "entropy": 0.2555816490203142, "clip_ratio/low_mean": 0.01687667949590832, "clip_ratio/low_min": 0.01687667949590832, "clip_ratio/high_mean": 0.02978548058308661, "clip_ratio/high_max": 0.02978548058308661, "clip_ratio/region_mean": 0.04666216007899493, "reward_total_mean": 0.8141774535179138, "reward_meter_mean": 0.9923086166381836, "reward_meter_std": 0.0017002333188429475, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8601362109184265, "reward_repeat_soft_std": 0.049438152462244034, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8141774535179138, "reward_total_composite_std": 0.007985465228557587} {"timestamp_utc": "2026-04-13T05:35:37Z", "mode": "train", "global_step": 2894, "epoch": 0.29070818684078353, "loss": 0.0191, "grad_norm": 6.323429584503174, "learning_rate": 1.2333333333333335e-06, "num_tokens": 5404802.0, "completions/mean_length": 152.0, "completions/min_length": 131.0, "completions/max_length": 175.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 152.0, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 175.0, "rewards/meter/mean": 0.5626073479652405, "rewards/meter/std": 0.310230016708374, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.946230411529541, "rewards/repeat_soft/std": 0.028350573033094406, "rewards/judge_quality/mean": 0.3137499690055847, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.5881713628768921, "rewards/total_composite/std": 0.1371031552553177, "reward": 0.5881713628768921, "reward_std": 0.1371031403541565, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11424139887094498, "sampling/sampling_logp_difference/max": 2.6258914470672607, "sampling/importance_sampling_ratio/min": 0.07237521559000015, "sampling/importance_sampling_ratio/mean": 0.9857437014579773, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37938980385661125, "clip_ratio/low_mean": 0.04657329712063074, "clip_ratio/low_min": 0.04657329712063074, "clip_ratio/high_mean": 0.0712202899158001, "clip_ratio/high_max": 0.0712202899158001, "clip_ratio/region_mean": 0.11779358703643084, "reward_total_mean": 0.5881713628768921, "reward_meter_mean": 0.5626073479652405, "reward_meter_std": 0.310230016708374, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.946230411529541, "reward_repeat_soft_std": 0.028350573033094406, "reward_judge_quality_mean": 0.3137499690055847, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.5881713628768921, "reward_total_composite_std": 0.1371031552553177} {"timestamp_utc": "2026-04-13T05:35:44Z", "mode": "train", "global_step": 2895, "epoch": 0.29080863887493724, "loss": -0.0119, "grad_norm": 11.668865203857422, "learning_rate": 1.2303030303030304e-06, "num_tokens": 5406268.0, "completions/mean_length": 32.25, "completions/min_length": 27.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.25, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.5478560924530029, "rewards/meter/std": 0.43793606758117676, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9615821838378906, "rewards/repeat_soft/std": 0.002596014179289341, "rewards/judge_quality/mean": 0.8612500429153442, "rewards/judge_quality/std": 0.16617010533809662, "rewards/total_composite/mean": 0.7510684728622437, "rewards/total_composite/std": 0.18273283541202545, "reward": 0.7510684728622437, "reward_std": 0.18273282051086426, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12710021436214447, "sampling/sampling_logp_difference/max": 1.2079224586486816, "sampling/importance_sampling_ratio/min": 0.2988174557685852, "sampling/importance_sampling_ratio/mean": 0.9925004839897156, "sampling/importance_sampling_ratio/max": 1.8543800115585327, "entropy": 0.5329459421336651, "clip_ratio/low_mean": 0.059461806900799274, "clip_ratio/low_min": 0.059461806900799274, "clip_ratio/high_mean": 0.08662227168679237, "clip_ratio/high_max": 0.08662227168679237, "clip_ratio/region_mean": 0.14608407858759165, "reward_total_mean": 0.7510684728622437, "reward_meter_mean": 0.5478560924530029, "reward_meter_std": 0.43793606758117676, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9615821838378906, "reward_repeat_soft_std": 0.002596014179289341, "reward_judge_quality_mean": 0.8612500429153442, "reward_judge_quality_std": 0.16617010533809662, "reward_total_composite_mean": 0.7510684728622437, "reward_total_composite_std": 0.18273283541202545} {"timestamp_utc": "2026-04-13T05:35:58Z", "mode": "train", "global_step": 2896, "epoch": 0.2909090909090909, "loss": -0.2385, "grad_norm": 1.5761741399765015, "learning_rate": 1.2272727272727274e-06, "num_tokens": 5409015.0, "completions/mean_length": 217.375, "completions/min_length": 161.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 175.2857208251953, "completions/min_terminated_length": 161.0, "completions/max_terminated_length": 187.0, "rewards/meter/mean": 0.8220894932746887, "rewards/meter/std": 0.31708192825317383, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.14880475401878357, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7299694418907166, "rewards/repeat_soft/std": 0.10498341917991638, "rewards/judge_quality/mean": 0.2574999928474426, "rewards/judge_quality/std": 0.155264750123024, "rewards/total_composite/mean": 0.631571888923645, "rewards/total_composite/std": 0.2679266929626465, "reward": 0.631571888923645, "reward_std": 0.2679266631603241, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06753071397542953, "sampling/sampling_logp_difference/max": 3.9080698490142822, "sampling/importance_sampling_ratio/min": 0.020079219713807106, "sampling/importance_sampling_ratio/mean": 1.0009121894836426, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2574837300926447, "clip_ratio/low_mean": 0.003881987649947405, "clip_ratio/low_min": 0.003881987649947405, "clip_ratio/high_mean": 0.04690955579280853, "clip_ratio/high_max": 0.04690955579280853, "clip_ratio/region_mean": 0.05079154344275594, "reward_total_mean": 0.631571888923645, "reward_meter_mean": 0.8220894932746887, "reward_meter_std": 0.31708192825317383, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.14880475401878357, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7299694418907166, "reward_repeat_soft_std": 0.10498341917991638, "reward_judge_quality_mean": 0.2574999928474426, "reward_judge_quality_std": 0.155264750123024, "reward_total_composite_mean": 0.631571888923645, "reward_total_composite_std": 0.2679266929626465} {"timestamp_utc": "2026-04-13T05:36:05Z", "mode": "train", "global_step": 2897, "epoch": 0.2910095429432446, "loss": 0.0177, "grad_norm": 7.278655529022217, "learning_rate": 1.2242424242424242e-06, "num_tokens": 5411319.0, "completions/mean_length": 117.0, "completions/min_length": 107.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.0, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.9027378559112549, "rewards/meter/std": 0.14862439036369324, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9413855075836182, "rewards/repeat_soft/std": 0.04622618854045868, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.7662455439567566, "rewards/total_composite/std": 0.06657056510448456, "reward": 0.7662455439567566, "reward_std": 0.06657055765390396, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08643970638513565, "sampling/sampling_logp_difference/max": 2.161360740661621, "sampling/importance_sampling_ratio/min": 0.11516829580068588, "sampling/importance_sampling_ratio/mean": 0.9984564781188965, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40016771480441093, "clip_ratio/low_mean": 0.03154719993472099, "clip_ratio/low_min": 0.03154719993472099, "clip_ratio/high_mean": 0.06562093738466501, "clip_ratio/high_max": 0.06562093738466501, "clip_ratio/region_mean": 0.097168137319386, "reward_total_mean": 0.7662455439567566, "reward_meter_mean": 0.9027378559112549, "reward_meter_std": 0.14862439036369324, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9413855075836182, "reward_repeat_soft_std": 0.04622618854045868, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.7662455439567566, "reward_total_composite_std": 0.06657056510448456} {"timestamp_utc": "2026-04-13T05:36:17Z", "mode": "train", "global_step": 2898, "epoch": 0.2911099949773983, "loss": -0.115, "grad_norm": 3.0322656631469727, "learning_rate": 1.2212121212121213e-06, "num_tokens": 5412919.0, "completions/mean_length": 104.0, "completions/min_length": 42.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 45.71428680419922, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.8036935329437256, "rewards/meter/std": 0.3034076690673828, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9661924839019775, "rewards/repeat_soft/std": 0.020722471177577972, "rewards/judge_quality/mean": 0.38875001668930054, "rewards/judge_quality/std": 0.13767844438552856, "rewards/total_composite/mean": 0.6399720311164856, "rewards/total_composite/std": 0.2921198010444641, "reward": 0.6399720311164856, "reward_std": 0.2921197712421417, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08506642282009125, "sampling/sampling_logp_difference/max": 2.404529094696045, "sampling/importance_sampling_ratio/min": 0.09030801057815552, "sampling/importance_sampling_ratio/mean": 1.000883936882019, "sampling/importance_sampling_ratio/max": 1.9063323736190796, "entropy": 0.3530581369996071, "clip_ratio/low_mean": 0.01944444514811039, "clip_ratio/low_min": 0.01944444514811039, "clip_ratio/high_mean": 0.05389811424538493, "clip_ratio/high_max": 0.05389811424538493, "clip_ratio/region_mean": 0.07334255939349532, "reward_total_mean": 0.6399720311164856, "reward_meter_mean": 0.8036935329437256, "reward_meter_std": 0.3034076690673828, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9661924839019775, "reward_repeat_soft_std": 0.020722471177577972, "reward_judge_quality_mean": 0.38875001668930054, "reward_judge_quality_std": 0.13767844438552856, "reward_total_composite_mean": 0.6399720311164856, "reward_total_composite_std": 0.2921198010444641} {"timestamp_utc": "2026-04-13T05:36:23Z", "mode": "train", "global_step": 2899, "epoch": 0.29121044701155196, "loss": -0.0084, "grad_norm": 8.433725357055664, "learning_rate": 1.2181818181818183e-06, "num_tokens": 5414638.0, "completions/mean_length": 54.875, "completions/min_length": 50.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.875, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.8540048003196716, "rewards/meter/std": 0.32766470313072205, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9433584809303284, "rewards/repeat_soft/std": 0.08112644404172897, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.7745130062103271, "rewards/total_composite/std": 0.15245278179645538, "reward": 0.7745130062103271, "reward_std": 0.15245279669761658, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09508245438337326, "sampling/sampling_logp_difference/max": 1.5056586265563965, "sampling/importance_sampling_ratio/min": 0.26938092708587646, "sampling/importance_sampling_ratio/mean": 1.0118833780288696, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5677182339131832, "clip_ratio/low_mean": 0.01715686358511448, "clip_ratio/low_min": 0.01715686358511448, "clip_ratio/high_mean": 0.09161421936005354, "clip_ratio/high_max": 0.09161421936005354, "clip_ratio/region_mean": 0.10877108294516802, "reward_total_mean": 0.7745130062103271, "reward_meter_mean": 0.8540048003196716, "reward_meter_std": 0.32766470313072205, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9433584809303284, "reward_repeat_soft_std": 0.08112644404172897, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.7745130062103271, "reward_total_composite_std": 0.15245278179645538} {"timestamp_utc": "2026-04-13T05:36:36Z", "mode": "train", "global_step": 2900, "epoch": 0.2913108990457057, "loss": -0.1702, "grad_norm": 1.5346919298171997, "learning_rate": 1.2151515151515154e-06, "num_tokens": 5416593.0, "completions/mean_length": 125.375, "completions/min_length": 66.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 70.14286041259766, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9404875040054321, "rewards/meter/std": 0.12826663255691528, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7721099853515625, "rewards/repeat_soft/std": 0.08947177231311798, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.13452960550785065, "rewards/total_composite/mean": 0.6970982551574707, "rewards/total_composite/std": 0.2817036211490631, "reward": 0.6970982551574707, "reward_std": 0.2817036211490631, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07450411468744278, "sampling/sampling_logp_difference/max": 1.9385466575622559, "sampling/importance_sampling_ratio/min": 0.14391295611858368, "sampling/importance_sampling_ratio/mean": 1.0039812326431274, "sampling/importance_sampling_ratio/max": 1.6441080570220947, "entropy": 0.3508993238210678, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06965372443664819, "clip_ratio/high_max": 0.06965372443664819, "clip_ratio/region_mean": 0.06965372443664819, "reward_total_mean": 0.6970982551574707, "reward_meter_mean": 0.9404875040054321, "reward_meter_std": 0.12826663255691528, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7721099853515625, "reward_repeat_soft_std": 0.08947177231311798, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.13452960550785065, "reward_total_composite_mean": 0.6970982551574707, "reward_total_composite_std": 0.2817036211490631} {"timestamp_utc": "2026-04-13T05:37:30Z", "mode": "eval", "global_step": 2900, "epoch": 0.2913108990457057, "eval_loss": NaN, "eval_runtime": 53.3988, "eval_samples_per_second": 1.498, "eval_steps_per_second": 0.187, "eval_num_tokens": 5416593.0, "eval_completions/mean_length": 101.9875, "eval_completions/min_length": 42.3, "eval_completions/max_length": 199.0, "eval_completions/clipped_ratio": 0.0125, "eval_completions/mean_terminated_length": 96.8357147216797, "eval_completions/min_terminated_length": 42.3, "eval_completions/max_terminated_length": 165.9, "eval_rewards/meter/mean": 0.9301794648170472, "eval_rewards/meter/std": 0.11335806455463171, "eval_rewards/count_adherence/mean": 0.9947916626930237, "eval_rewards/count_adherence/std": 0.01473139189183712, "eval_rewards/hard_gate/mean": 0.9875, "eval_rewards/hard_gate/std": 0.03535533845424652, "eval_rewards/repeat_soft/mean": 0.9122387826442718, "eval_rewards/repeat_soft/std": 0.0857144195586443, "eval_rewards/judge_quality/mean": 0.4233749985694885, "eval_rewards/judge_quality/std": 0.1400534668006003, "eval_rewards/total_composite/mean": 0.7818772494792938, "eval_rewards/total_composite/std": 0.08786705583333969, "eval_reward": 0.7818772494792938, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.040445775166153906, "eval_sampling/sampling_logp_difference/max": 0.9412353992462158, "eval_sampling/importance_sampling_ratio/min": 0.4001949280500412, "eval_sampling/importance_sampling_ratio/mean": 1.0057626962661743, "eval_sampling/importance_sampling_ratio/max": 1.291041898727417, "eval_entropy": 0.38925869166851046, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7818772494792938, "eval_reward_meter_mean": 0.9301794648170472, "eval_reward_meter_std": 0.11335806455463171, "eval_reward_count_adherence_mean": 0.9947916626930237, "eval_reward_count_adherence_std": 0.01473139189183712, "eval_reward_hard_gate_mean": 0.9875, "eval_reward_hard_gate_std": 0.03535533845424652, "eval_reward_repeat_soft_mean": 0.9122387826442718, "eval_reward_repeat_soft_std": 0.0857144195586443, "eval_reward_judge_quality_mean": 0.4233749985694885, "eval_reward_judge_quality_std": 0.1400534668006003, "eval_reward_total_composite_mean": 0.7818772494792938, "eval_reward_total_composite_std": 0.08786705583333969} {"timestamp_utc": "2026-04-13T05:37:41Z", "mode": "train", "global_step": 2901, "epoch": 0.2914113510798594, "loss": 0.0505, "grad_norm": 9.682042121887207, "learning_rate": 1.2121212121212122e-06, "num_tokens": 5418365.0, "completions/mean_length": 56.5, "completions/min_length": 50.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.5, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9873770475387573, "rewards/meter/std": 0.017035437747836113, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.946733832359314, "rewards/repeat_soft/std": 0.048090461641550064, "rewards/judge_quality/mean": 0.5575000047683716, "rewards/judge_quality/std": 0.19955310225486755, "rewards/total_composite/mean": 0.8562430739402771, "rewards/total_composite/std": 0.05681659281253815, "reward": 0.8562430739402771, "reward_std": 0.056816600263118744, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07306715101003647, "sampling/sampling_logp_difference/max": 1.066986083984375, "sampling/importance_sampling_ratio/min": 0.34567826986312866, "sampling/importance_sampling_ratio/mean": 1.0162956714630127, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3931319788098335, "clip_ratio/low_mean": 0.03663488803431392, "clip_ratio/low_min": 0.03663488803431392, "clip_ratio/high_mean": 0.01870414661243558, "clip_ratio/high_max": 0.01870414661243558, "clip_ratio/region_mean": 0.055339034646749496, "reward_total_mean": 0.8562430739402771, "reward_meter_mean": 0.9873770475387573, "reward_meter_std": 0.017035437747836113, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.946733832359314, "reward_repeat_soft_std": 0.048090461641550064, "reward_judge_quality_mean": 0.5575000047683716, "reward_judge_quality_std": 0.19955310225486755, "reward_total_composite_mean": 0.8562430739402771, "reward_total_composite_std": 0.05681659281253815} {"timestamp_utc": "2026-04-13T05:37:48Z", "mode": "train", "global_step": 2902, "epoch": 0.29151180311401304, "loss": 0.0047, "grad_norm": 10.880475044250488, "learning_rate": 1.2090909090909092e-06, "num_tokens": 5420102.0, "completions/mean_length": 57.125, "completions/min_length": 54.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.125, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9900362491607666, "rewards/meter/std": 0.0047183558344841, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9477518200874329, "rewards/repeat_soft/std": 0.03247014805674553, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.1011011004447937, "rewards/total_composite/mean": 0.8320415019989014, "rewards/total_composite/std": 0.0295230932533741, "reward": 0.8320415019989014, "reward_std": 0.029523082077503204, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08053744584321976, "sampling/sampling_logp_difference/max": 1.365915298461914, "sampling/importance_sampling_ratio/min": 0.2551470398902893, "sampling/importance_sampling_ratio/mean": 0.9996411204338074, "sampling/importance_sampling_ratio/max": 1.9628971815109253, "entropy": 0.4490970969200134, "clip_ratio/low_mean": 0.06370993191376328, "clip_ratio/low_min": 0.06370993191376328, "clip_ratio/high_mean": 0.008620689623057842, "clip_ratio/high_max": 0.008620689623057842, "clip_ratio/region_mean": 0.07233062153682113, "reward_total_mean": 0.8320415019989014, "reward_meter_mean": 0.9900362491607666, "reward_meter_std": 0.0047183558344841, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9477518200874329, "reward_repeat_soft_std": 0.03247014805674553, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.1011011004447937, "reward_total_composite_mean": 0.8320415019989014, "reward_total_composite_std": 0.0295230932533741} {"timestamp_utc": "2026-04-13T05:37:56Z", "mode": "train", "global_step": 2903, "epoch": 0.29161225514816674, "loss": 0.0139, "grad_norm": 5.009574890136719, "learning_rate": 1.206060606060606e-06, "num_tokens": 5422113.0, "completions/mean_length": 93.375, "completions/min_length": 90.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.375, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.8894660472869873, "rewards/meter/std": 0.22354136407375336, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8309808969497681, "rewards/repeat_soft/std": 0.06553151458501816, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.7781078219413757, "rewards/total_composite/std": 0.045134514570236206, "reward": 0.7781078219413757, "reward_std": 0.04513451084494591, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05223499611020088, "sampling/sampling_logp_difference/max": 1.3995425701141357, "sampling/importance_sampling_ratio/min": 0.24670977890491486, "sampling/importance_sampling_ratio/mean": 1.0003076791763306, "sampling/importance_sampling_ratio/max": 1.8762271404266357, "entropy": 0.273335762321949, "clip_ratio/low_mean": 0.012013730127364397, "clip_ratio/low_min": 0.012013730127364397, "clip_ratio/high_mean": 0.03621704061515629, "clip_ratio/high_max": 0.03621704061515629, "clip_ratio/region_mean": 0.04823077074252069, "reward_total_mean": 0.7781078219413757, "reward_meter_mean": 0.8894660472869873, "reward_meter_std": 0.22354136407375336, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8309808969497681, "reward_repeat_soft_std": 0.06553151458501816, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.7781078219413757, "reward_total_composite_std": 0.045134514570236206} {"timestamp_utc": "2026-04-13T05:38:09Z", "mode": "train", "global_step": 2904, "epoch": 0.29171270718232045, "loss": -0.0993, "grad_norm": 1.6464577913284302, "learning_rate": 1.2030303030303031e-06, "num_tokens": 5423572.0, "completions/mean_length": 90.375, "completions/min_length": 25.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 30.142858505249023, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.8657987117767334, "rewards/meter/std": 0.30725574493408203, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9601136445999146, "rewards/repeat_soft/std": 0.006749649066478014, "rewards/judge_quality/mean": 0.3149999976158142, "rewards/judge_quality/std": 0.15004761517047882, "rewards/total_composite/mean": 0.6910791397094727, "rewards/total_composite/std": 0.2826089560985565, "reward": 0.6910791397094727, "reward_std": 0.2826089560985565, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10562553256750107, "sampling/sampling_logp_difference/max": 1.1542930603027344, "sampling/importance_sampling_ratio/min": 0.31528031826019287, "sampling/importance_sampling_ratio/mean": 1.0091198682785034, "sampling/importance_sampling_ratio/max": 1.8816089630126953, "entropy": 0.5196858271956444, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09019849286414683, "clip_ratio/high_max": 0.09019849286414683, "clip_ratio/region_mean": 0.09019849286414683, "reward_total_mean": 0.6910791397094727, "reward_meter_mean": 0.8657987117767334, "reward_meter_std": 0.30725574493408203, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9601136445999146, "reward_repeat_soft_std": 0.006749649066478014, "reward_judge_quality_mean": 0.3149999976158142, "reward_judge_quality_std": 0.15004761517047882, "reward_total_composite_mean": 0.6910791397094727, "reward_total_composite_std": 0.2826089560985565} {"timestamp_utc": "2026-04-13T05:38:21Z", "mode": "train", "global_step": 2905, "epoch": 0.29181315921647416, "loss": -0.1922, "grad_norm": 2.6116504669189453, "learning_rate": 1.2000000000000002e-06, "num_tokens": 5425734.0, "completions/mean_length": 162.25, "completions/min_length": 104.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 112.28572082519531, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.7185829877853394, "rewards/meter/std": 0.24589303135871887, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.981975793838501, "rewards/repeat_soft/std": 0.005715638864785433, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.2988907992839813, "rewards/total_composite/mean": 0.6644572019577026, "rewards/total_composite/std": 0.29026639461517334, "reward": 0.6644572019577026, "reward_std": 0.29026636481285095, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10230498015880585, "sampling/sampling_logp_difference/max": 1.8811465501785278, "sampling/importance_sampling_ratio/min": 0.15241526067256927, "sampling/importance_sampling_ratio/mean": 0.9986600279808044, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36353152617812157, "clip_ratio/low_mean": 0.022633075714111328, "clip_ratio/low_min": 0.022633075714111328, "clip_ratio/high_mean": 0.07099988590925932, "clip_ratio/high_max": 0.07099988590925932, "clip_ratio/region_mean": 0.09363296162337065, "reward_total_mean": 0.6644572019577026, "reward_meter_mean": 0.7185829877853394, "reward_meter_std": 0.24589303135871887, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.981975793838501, "reward_repeat_soft_std": 0.005715638864785433, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.2988907992839813, "reward_total_composite_mean": 0.6644572019577026, "reward_total_composite_std": 0.29026639461517334} {"timestamp_utc": "2026-04-13T05:38:28Z", "mode": "train", "global_step": 2906, "epoch": 0.2919136112506278, "loss": 0.0422, "grad_norm": 7.9846086502075195, "learning_rate": 1.196969696969697e-06, "num_tokens": 5427517.0, "completions/mean_length": 51.875, "completions/min_length": 48.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.875, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9933475852012634, "rewards/meter/std": 0.004348048008978367, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9663381576538086, "rewards/repeat_soft/std": 0.01634363643825054, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8263902068138123, "rewards/total_composite/std": 0.004603346344083548, "reward": 0.8263902068138123, "reward_std": 0.0046033407561481, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0817505270242691, "sampling/sampling_logp_difference/max": 1.5852065086364746, "sampling/importance_sampling_ratio/min": 0.2049054652452469, "sampling/importance_sampling_ratio/mean": 0.9952142834663391, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.33931877091526985, "clip_ratio/low_mean": 0.027591857127845287, "clip_ratio/low_min": 0.027591857127845287, "clip_ratio/high_mean": 0.06113483523949981, "clip_ratio/high_max": 0.06113483523949981, "clip_ratio/region_mean": 0.0887266923673451, "reward_total_mean": 0.8263902068138123, "reward_meter_mean": 0.9933475852012634, "reward_meter_std": 0.004348048008978367, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9663381576538086, "reward_repeat_soft_std": 0.01634363643825054, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8263902068138123, "reward_total_composite_std": 0.004603346344083548} {"timestamp_utc": "2026-04-13T05:38:35Z", "mode": "train", "global_step": 2907, "epoch": 0.2920140632847815, "loss": -0.0382, "grad_norm": 7.523456573486328, "learning_rate": 1.193939393939394e-06, "num_tokens": 5429046.0, "completions/mean_length": 29.125, "completions/min_length": 24.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.125, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.8820362091064453, "rewards/meter/std": 0.2396921068429947, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9528409242630005, "rewards/repeat_soft/std": 0.027320031076669693, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.7892003655433655, "rewards/total_composite/std": 0.12683533132076263, "reward": 0.7892003655433655, "reward_std": 0.12683531641960144, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07384398579597473, "sampling/sampling_logp_difference/max": 1.4513583183288574, "sampling/importance_sampling_ratio/min": 0.2992078363895416, "sampling/importance_sampling_ratio/mean": 1.0081719160079956, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3397714588791132, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.049341903533786535, "clip_ratio/high_max": 0.049341903533786535, "clip_ratio/region_mean": 0.049341903533786535, "reward_total_mean": 0.7892003655433655, "reward_meter_mean": 0.8820362091064453, "reward_meter_std": 0.2396921068429947, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9528409242630005, "reward_repeat_soft_std": 0.027320031076669693, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.7892003655433655, "reward_total_composite_std": 0.12683533132076263} {"timestamp_utc": "2026-04-13T05:38:42Z", "mode": "train", "global_step": 2908, "epoch": 0.29211451531893523, "loss": 0.0234, "grad_norm": 7.837900161743164, "learning_rate": 1.190909090909091e-06, "num_tokens": 5430943.0, "completions/mean_length": 67.125, "completions/min_length": 65.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.125, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9899146556854248, "rewards/meter/std": 0.006183892022818327, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9430147409439087, "rewards/repeat_soft/std": 0.032636821269989014, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8168880939483643, "rewards/total_composite/std": 0.00380168529227376, "reward": 0.8168880939483643, "reward_std": 0.003801694605499506, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07289174944162369, "sampling/sampling_logp_difference/max": 1.2399311065673828, "sampling/importance_sampling_ratio/min": 0.28940415382385254, "sampling/importance_sampling_ratio/mean": 1.0018792152404785, "sampling/importance_sampling_ratio/max": 1.8264843225479126, "entropy": 0.3941251300275326, "clip_ratio/low_mean": 0.020503953099250793, "clip_ratio/low_min": 0.020503953099250793, "clip_ratio/high_mean": 0.05388102610595524, "clip_ratio/high_max": 0.05388102610595524, "clip_ratio/region_mean": 0.07438497920520604, "reward_total_mean": 0.8168880939483643, "reward_meter_mean": 0.9899146556854248, "reward_meter_std": 0.006183892022818327, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9430147409439087, "reward_repeat_soft_std": 0.032636821269989014, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8168880939483643, "reward_total_composite_std": 0.00380168529227376} {"timestamp_utc": "2026-04-13T05:38:50Z", "mode": "train", "global_step": 2909, "epoch": 0.2922149673530889, "loss": 0.0395, "grad_norm": 5.877314567565918, "learning_rate": 1.187878787878788e-06, "num_tokens": 5433630.0, "completions/mean_length": 146.875, "completions/min_length": 132.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 146.875, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.9868953227996826, "rewards/meter/std": 0.01081900205463171, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8889329433441162, "rewards/repeat_soft/std": 0.03976831212639809, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.7861211895942688, "rewards/total_composite/std": 0.03337760269641876, "reward": 0.7861211895942688, "reward_std": 0.03337761014699936, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08770257234573364, "sampling/sampling_logp_difference/max": 3.0435051918029785, "sampling/importance_sampling_ratio/min": 0.04766751453280449, "sampling/importance_sampling_ratio/mean": 0.996643602848053, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37347306311130524, "clip_ratio/low_mean": 0.026805629953742027, "clip_ratio/low_min": 0.026805629953742027, "clip_ratio/high_mean": 0.058771153911948204, "clip_ratio/high_max": 0.058771153911948204, "clip_ratio/region_mean": 0.08557678386569023, "reward_total_mean": 0.7861211895942688, "reward_meter_mean": 0.9868953227996826, "reward_meter_std": 0.01081900205463171, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8889329433441162, "reward_repeat_soft_std": 0.03976831212639809, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.7861211895942688, "reward_total_composite_std": 0.03337760269641876} {"timestamp_utc": "2026-04-13T05:38:58Z", "mode": "train", "global_step": 2910, "epoch": 0.2923154193872426, "loss": 0.014, "grad_norm": 7.207554340362549, "learning_rate": 1.184848484848485e-06, "num_tokens": 5435702.0, "completions/mean_length": 100.0, "completions/min_length": 91.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.0, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.9954690933227539, "rewards/meter/std": 0.0009317319490946829, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.967071533203125, "rewards/repeat_soft/std": 0.0132428715005517, "rewards/judge_quality/mean": 0.5699999928474426, "rewards/judge_quality/std": 0.16035676002502441, "rewards/total_composite/mean": 0.8656682372093201, "rewards/total_composite/std": 0.047562770545482635, "reward": 0.8656682372093201, "reward_std": 0.04756275564432144, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11049548536539078, "sampling/sampling_logp_difference/max": 2.15586519241333, "sampling/importance_sampling_ratio/min": 0.11580295860767365, "sampling/importance_sampling_ratio/mean": 0.9963944554328918, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43982164561748505, "clip_ratio/low_mean": 0.03907509194687009, "clip_ratio/low_min": 0.03907509194687009, "clip_ratio/high_mean": 0.053153956308960915, "clip_ratio/high_max": 0.053153956308960915, "clip_ratio/region_mean": 0.092229048255831, "reward_total_mean": 0.8656682372093201, "reward_meter_mean": 0.9954690933227539, "reward_meter_std": 0.0009317319490946829, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.967071533203125, "reward_repeat_soft_std": 0.0132428715005517, "reward_judge_quality_mean": 0.5699999928474426, "reward_judge_quality_std": 0.16035676002502441, "reward_total_composite_mean": 0.8656682372093201, "reward_total_composite_std": 0.047562770545482635} {"timestamp_utc": "2026-04-13T05:39:06Z", "mode": "train", "global_step": 2911, "epoch": 0.2924158714213963, "loss": 0.038, "grad_norm": 7.633604049682617, "learning_rate": 1.181818181818182e-06, "num_tokens": 5437880.0, "completions/mean_length": 93.25, "completions/min_length": 87.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.25, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.9922110438346863, "rewards/meter/std": 0.0030345183331519365, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9273275136947632, "rewards/repeat_soft/std": 0.029938776046037674, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.9089776873588562, "rewards/total_composite/std": 0.07577270269393921, "reward": 0.9089776873588562, "reward_std": 0.07577269524335861, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09019917249679565, "sampling/sampling_logp_difference/max": 1.620435357093811, "sampling/importance_sampling_ratio/min": 0.19781257212162018, "sampling/importance_sampling_ratio/mean": 0.9952515363693237, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44222957268357277, "clip_ratio/low_mean": 0.02967909909784794, "clip_ratio/low_min": 0.02967909909784794, "clip_ratio/high_mean": 0.06049030926078558, "clip_ratio/high_max": 0.06049030926078558, "clip_ratio/region_mean": 0.09016940835863352, "reward_total_mean": 0.9089776873588562, "reward_meter_mean": 0.9922110438346863, "reward_meter_std": 0.0030345183331519365, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9273275136947632, "reward_repeat_soft_std": 0.029938776046037674, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.9089776873588562, "reward_total_composite_std": 0.07577270269393921} {"timestamp_utc": "2026-04-13T05:39:14Z", "mode": "train", "global_step": 2912, "epoch": 0.29251632345554995, "loss": 0.0721, "grad_norm": 4.331712245941162, "learning_rate": 1.1787878787878788e-06, "num_tokens": 5440470.0, "completions/mean_length": 132.75, "completions/min_length": 108.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.75, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.96807861328125, "rewards/meter/std": 0.021306617185473442, "rewards/count_adherence/mean": 0.8500000238418579, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.777576208114624, "rewards/repeat_soft/std": 0.07237795740365982, "rewards/judge_quality/mean": 0.29249998927116394, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7286429405212402, "rewards/total_composite/std": 0.03321171551942825, "reward": 0.7286429405212402, "reward_std": 0.033211711794137955, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05485259369015694, "sampling/sampling_logp_difference/max": 1.7713546752929688, "sampling/importance_sampling_ratio/min": 0.17010240256786346, "sampling/importance_sampling_ratio/mean": 1.0046699047088623, "sampling/importance_sampling_ratio/max": 1.7720260620117188, "entropy": 0.27617561630904675, "clip_ratio/low_mean": 0.03047918830998242, "clip_ratio/low_min": 0.03047918830998242, "clip_ratio/high_mean": 0.020494621247053146, "clip_ratio/high_max": 0.020494621247053146, "clip_ratio/region_mean": 0.050973809557035565, "reward_total_mean": 0.7286429405212402, "reward_meter_mean": 0.96807861328125, "reward_meter_std": 0.021306617185473442, "reward_count_adherence_mean": 0.8500000238418579, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.777576208114624, "reward_repeat_soft_std": 0.07237795740365982, "reward_judge_quality_mean": 0.29249998927116394, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7286429405212402, "reward_total_composite_std": 0.03321171551942825} {"timestamp_utc": "2026-04-13T05:39:21Z", "mode": "train", "global_step": 2913, "epoch": 0.29261677548970366, "loss": 0.0107, "grad_norm": 17.911317825317383, "learning_rate": 1.1757575757575759e-06, "num_tokens": 5441946.0, "completions/mean_length": 24.5, "completions/min_length": 22.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.5, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.9828400015830994, "rewards/meter/std": 0.006107419263571501, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8858115673065186, "rewards/repeat_soft/std": 0.06481926888227463, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.08166787773370743, "rewards/total_composite/mean": 0.7952341437339783, "rewards/total_composite/std": 0.027183745056390762, "reward": 0.7952341437339783, "reward_std": 0.027183735743165016, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09505486488342285, "sampling/sampling_logp_difference/max": 1.0315008163452148, "sampling/importance_sampling_ratio/min": 0.35647156834602356, "sampling/importance_sampling_ratio/mean": 0.9835548996925354, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4240809194743633, "clip_ratio/low_mean": 0.014807692263275385, "clip_ratio/low_min": 0.014807692263275385, "clip_ratio/high_mean": 0.08130434714257717, "clip_ratio/high_max": 0.08130434714257717, "clip_ratio/region_mean": 0.09611203940585256, "reward_total_mean": 0.7952341437339783, "reward_meter_mean": 0.9828400015830994, "reward_meter_std": 0.006107419263571501, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8858115673065186, "reward_repeat_soft_std": 0.06481926888227463, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.08166787773370743, "reward_total_composite_mean": 0.7952341437339783, "reward_total_composite_std": 0.027183745056390762} {"timestamp_utc": "2026-04-13T05:39:29Z", "mode": "train", "global_step": 2914, "epoch": 0.29271722752385737, "loss": 0.0297, "grad_norm": 6.613576889038086, "learning_rate": 1.172727272727273e-06, "num_tokens": 5444376.0, "completions/mean_length": 115.75, "completions/min_length": 110.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.75, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.9914754629135132, "rewards/meter/std": 0.003348794300109148, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8748539686203003, "rewards/repeat_soft/std": 0.034370552748441696, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.24842002987861633, "rewards/total_composite/mean": 0.850524365901947, "rewards/total_composite/std": 0.0748961865901947, "reward": 0.850524365901947, "reward_std": 0.0748961865901947, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08383754640817642, "sampling/sampling_logp_difference/max": 1.5667824745178223, "sampling/importance_sampling_ratio/min": 0.20871564745903015, "sampling/importance_sampling_ratio/mean": 1.0013054609298706, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4163365922868252, "clip_ratio/low_mean": 0.049232271034270525, "clip_ratio/low_min": 0.049232271034270525, "clip_ratio/high_mean": 0.038534137420356274, "clip_ratio/high_max": 0.038534137420356274, "clip_ratio/region_mean": 0.0877664084546268, "reward_total_mean": 0.850524365901947, "reward_meter_mean": 0.9914754629135132, "reward_meter_std": 0.003348794300109148, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8748539686203003, "reward_repeat_soft_std": 0.034370552748441696, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.24842002987861633, "reward_total_composite_mean": 0.850524365901947, "reward_total_composite_std": 0.0748961865901947} {"timestamp_utc": "2026-04-13T05:39:41Z", "mode": "train", "global_step": 2915, "epoch": 0.292817679558011, "loss": -0.1651, "grad_norm": 1.8962608575820923, "learning_rate": 1.1696969696969697e-06, "num_tokens": 5446324.0, "completions/mean_length": 124.5, "completions/min_length": 61.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 69.14286041259766, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9216798543930054, "rewards/meter/std": 0.06634940952062607, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9173114895820618, "rewards/repeat_soft/std": 0.07301369309425354, "rewards/judge_quality/mean": 0.44875001907348633, "rewards/judge_quality/std": 0.2105392962694168, "rewards/total_composite/mean": 0.7119457125663757, "rewards/total_composite/std": 0.2902553379535675, "reward": 0.7119457125663757, "reward_std": 0.29025527834892273, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09677641093730927, "sampling/sampling_logp_difference/max": 1.6936006546020508, "sampling/importance_sampling_ratio/min": 0.1838563233613968, "sampling/importance_sampling_ratio/mean": 1.0003324747085571, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3874734193086624, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08232529740780592, "clip_ratio/high_max": 0.08232529740780592, "clip_ratio/region_mean": 0.08232529740780592, "reward_total_mean": 0.7119457125663757, "reward_meter_mean": 0.9216798543930054, "reward_meter_std": 0.06634940952062607, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9173114895820618, "reward_repeat_soft_std": 0.07301369309425354, "reward_judge_quality_mean": 0.44875001907348633, "reward_judge_quality_std": 0.2105392962694168, "reward_total_composite_mean": 0.7119457125663757, "reward_total_composite_std": 0.2902553379535675} {"timestamp_utc": "2026-04-13T05:39:49Z", "mode": "train", "global_step": 2916, "epoch": 0.29291813159216473, "loss": -0.0125, "grad_norm": 6.946169853210449, "learning_rate": 1.1666666666666668e-06, "num_tokens": 5448589.0, "completions/mean_length": 115.125, "completions/min_length": 107.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.125, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.8219122290611267, "rewards/meter/std": 0.32749873399734497, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8986859917640686, "rewards/repeat_soft/std": 0.03047974407672882, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.7544791102409363, "rewards/total_composite/std": 0.14281955361366272, "reward": 0.7544791102409363, "reward_std": 0.14281955361366272, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0667281374335289, "sampling/sampling_logp_difference/max": 1.79634428024292, "sampling/importance_sampling_ratio/min": 0.16590426862239838, "sampling/importance_sampling_ratio/mean": 1.00943922996521, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3623035401105881, "clip_ratio/low_mean": 0.009345794096589088, "clip_ratio/low_min": 0.009345794096589088, "clip_ratio/high_mean": 0.06981351459398866, "clip_ratio/high_max": 0.06981351459398866, "clip_ratio/region_mean": 0.07915930869057775, "reward_total_mean": 0.7544791102409363, "reward_meter_mean": 0.8219122290611267, "reward_meter_std": 0.32749873399734497, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8986859917640686, "reward_repeat_soft_std": 0.03047974407672882, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.7544791102409363, "reward_total_composite_std": 0.14281955361366272} {"timestamp_utc": "2026-04-13T05:39:57Z", "mode": "train", "global_step": 2917, "epoch": 0.29301858362631844, "loss": 0.0092, "grad_norm": 16.560300827026367, "learning_rate": 1.1636363636363638e-06, "num_tokens": 5450473.0, "completions/mean_length": 62.5, "completions/min_length": 58.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.5, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9854568243026733, "rewards/meter/std": 0.006389974150806665, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9551903605461121, "rewards/repeat_soft/std": 0.04898300766944885, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.8194745779037476, "rewards/total_composite/std": 0.005337316542863846, "reward": 0.8194745779037476, "reward_std": 0.005337311886250973, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0902002677321434, "sampling/sampling_logp_difference/max": 1.7439818382263184, "sampling/importance_sampling_ratio/min": 0.17482289671897888, "sampling/importance_sampling_ratio/mean": 0.9983859062194824, "sampling/importance_sampling_ratio/max": 1.8926292657852173, "entropy": 0.5025358386337757, "clip_ratio/low_mean": 0.054257793352007866, "clip_ratio/low_min": 0.054257793352007866, "clip_ratio/high_mean": 0.03801744244992733, "clip_ratio/high_max": 0.03801744244992733, "clip_ratio/region_mean": 0.0922752358019352, "reward_total_mean": 0.8194745779037476, "reward_meter_mean": 0.9854568243026733, "reward_meter_std": 0.006389974150806665, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9551903605461121, "reward_repeat_soft_std": 0.04898300766944885, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.8194745779037476, "reward_total_composite_std": 0.005337316542863846} {"timestamp_utc": "2026-04-13T05:40:05Z", "mode": "train", "global_step": 2918, "epoch": 0.29311903566047215, "loss": 0.0316, "grad_norm": 10.687793731689453, "learning_rate": 1.1606060606060607e-06, "num_tokens": 5452094.0, "completions/mean_length": 29.625, "completions/min_length": 27.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.625, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9914635419845581, "rewards/meter/std": 0.0042123072780668736, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9341965913772583, "rewards/repeat_soft/std": 0.05753038451075554, "rewards/judge_quality/mean": 0.32499998807907104, "rewards/judge_quality/std": 0.1035098284482956, "rewards/total_composite/mean": 0.7870782613754272, "rewards/total_composite/std": 0.033200256526470184, "reward": 0.7870782613754272, "reward_std": 0.033200234174728394, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06488898396492004, "sampling/sampling_logp_difference/max": 0.6315140724182129, "sampling/importance_sampling_ratio/min": 0.5466369986534119, "sampling/importance_sampling_ratio/mean": 1.0310909748077393, "sampling/importance_sampling_ratio/max": 1.880455493927002, "entropy": 0.433343093842268, "clip_ratio/low_mean": 0.055683840066194534, "clip_ratio/low_min": 0.055683840066194534, "clip_ratio/high_mean": 0.028594771400094032, "clip_ratio/high_max": 0.028594771400094032, "clip_ratio/region_mean": 0.08427861146628857, "reward_total_mean": 0.7870782613754272, "reward_meter_mean": 0.9914635419845581, "reward_meter_std": 0.0042123072780668736, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9341965913772583, "reward_repeat_soft_std": 0.05753038451075554, "reward_judge_quality_mean": 0.32499998807907104, "reward_judge_quality_std": 0.1035098284482956, "reward_total_composite_mean": 0.7870782613754272, "reward_total_composite_std": 0.033200256526470184} {"timestamp_utc": "2026-04-13T05:40:12Z", "mode": "train", "global_step": 2919, "epoch": 0.2932194876946258, "loss": 0.0108, "grad_norm": 9.331555366516113, "learning_rate": 1.1575757575757577e-06, "num_tokens": 5453844.0, "completions/mean_length": 54.75, "completions/min_length": 50.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.75, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9825541973114014, "rewards/meter/std": 0.02293606661260128, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9186885356903076, "rewards/repeat_soft/std": 0.03770734742283821, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.8310182094573975, "rewards/total_composite/std": 0.04544555023312569, "reward": 0.8310182094573975, "reward_std": 0.04544554278254509, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08909907191991806, "sampling/sampling_logp_difference/max": 1.8867765665054321, "sampling/importance_sampling_ratio/min": 0.15155956149101257, "sampling/importance_sampling_ratio/mean": 0.9987015128135681, "sampling/importance_sampling_ratio/max": 1.7285888195037842, "entropy": 0.39904771000146866, "clip_ratio/low_mean": 0.05790414800867438, "clip_ratio/low_min": 0.05790414800867438, "clip_ratio/high_mean": 0.01785714365541935, "clip_ratio/high_max": 0.01785714365541935, "clip_ratio/region_mean": 0.07576129166409373, "reward_total_mean": 0.8310182094573975, "reward_meter_mean": 0.9825541973114014, "reward_meter_std": 0.02293606661260128, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9186885356903076, "reward_repeat_soft_std": 0.03770734742283821, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.8310182094573975, "reward_total_composite_std": 0.04544555023312569} {"timestamp_utc": "2026-04-13T05:40:20Z", "mode": "train", "global_step": 2920, "epoch": 0.2933199397287795, "loss": 0.0372, "grad_norm": 11.32181167602539, "learning_rate": 1.1545454545454545e-06, "num_tokens": 5455855.0, "completions/mean_length": 84.375, "completions/min_length": 77.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.375, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.9904906749725342, "rewards/meter/std": 0.0033858255483210087, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8934662938117981, "rewards/repeat_soft/std": 0.045697156339883804, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.8046923875808716, "rewards/total_composite/std": 0.017710011452436447, "reward": 0.8046923875808716, "reward_std": 0.017710011452436447, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09385846555233002, "sampling/sampling_logp_difference/max": 1.4055042266845703, "sampling/importance_sampling_ratio/min": 0.2452433556318283, "sampling/importance_sampling_ratio/mean": 0.9971371293067932, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37528448179364204, "clip_ratio/low_mean": 0.014630642253905535, "clip_ratio/low_min": 0.014630642253905535, "clip_ratio/high_mean": 0.06360393203794956, "clip_ratio/high_max": 0.06360393203794956, "clip_ratio/region_mean": 0.0782345742918551, "reward_total_mean": 0.8046923875808716, "reward_meter_mean": 0.9904906749725342, "reward_meter_std": 0.0033858255483210087, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8934662938117981, "reward_repeat_soft_std": 0.045697156339883804, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.8046923875808716, "reward_total_composite_std": 0.017710011452436447} {"timestamp_utc": "2026-04-13T05:40:32Z", "mode": "train", "global_step": 2921, "epoch": 0.2934203917629332, "loss": -0.1038, "grad_norm": 1.6688786745071411, "learning_rate": 1.1515151515151516e-06, "num_tokens": 5457374.0, "completions/mean_length": 161.875, "completions/min_length": 42.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 45.16666793823242, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.6741542816162109, "rewards/meter/std": 0.38523682951927185, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.981981098651886, "rewards/repeat_soft/std": 0.01757725141942501, "rewards/judge_quality/mean": 0.3662499785423279, "rewards/judge_quality/std": 0.28535380959510803, "rewards/total_composite/mean": 0.5904441475868225, "rewards/total_composite/std": 0.3747292757034302, "reward": 0.5904441475868225, "reward_std": 0.3747292757034302, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12327657639980316, "sampling/sampling_logp_difference/max": 1.479170322418213, "sampling/importance_sampling_ratio/min": 0.22782664000988007, "sampling/importance_sampling_ratio/mean": 1.0026977062225342, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3890044316649437, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09409521240741014, "clip_ratio/high_max": 0.09409521240741014, "clip_ratio/region_mean": 0.09409521240741014, "reward_total_mean": 0.5904441475868225, "reward_meter_mean": 0.6741542816162109, "reward_meter_std": 0.38523682951927185, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.981981098651886, "reward_repeat_soft_std": 0.01757725141942501, "reward_judge_quality_mean": 0.3662499785423279, "reward_judge_quality_std": 0.28535380959510803, "reward_total_composite_mean": 0.5904441475868225, "reward_total_composite_std": 0.3747292757034302} {"timestamp_utc": "2026-04-13T05:40:45Z", "mode": "train", "global_step": 2922, "epoch": 0.29352084379708687, "loss": -0.0774, "grad_norm": 1.0857019424438477, "learning_rate": 1.1484848484848486e-06, "num_tokens": 5458797.0, "completions/mean_length": 82.875, "completions/min_length": 18.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 21.571430206298828, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.8119586110115051, "rewards/meter/std": 0.3421173095703125, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9493010640144348, "rewards/repeat_soft/std": 0.05380989611148834, "rewards/judge_quality/mean": 0.3399999737739563, "rewards/judge_quality/std": 0.15052290260791779, "rewards/total_composite/mean": 0.6789364814758301, "rewards/total_composite/std": 0.27818384766578674, "reward": 0.6789364814758301, "reward_std": 0.27818384766578674, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06782343238592148, "sampling/sampling_logp_difference/max": 0.804232120513916, "sampling/importance_sampling_ratio/min": 0.44743138551712036, "sampling/importance_sampling_ratio/mean": 0.9930977821350098, "sampling/importance_sampling_ratio/max": 1.7214958667755127, "entropy": 0.3558388203382492, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08389406185597181, "clip_ratio/high_max": 0.08389406185597181, "clip_ratio/region_mean": 0.08389406185597181, "reward_total_mean": 0.6789364814758301, "reward_meter_mean": 0.8119586110115051, "reward_meter_std": 0.3421173095703125, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9493010640144348, "reward_repeat_soft_std": 0.05380989611148834, "reward_judge_quality_mean": 0.3399999737739563, "reward_judge_quality_std": 0.15052290260791779, "reward_total_composite_mean": 0.6789364814758301, "reward_total_composite_std": 0.27818384766578674} {"timestamp_utc": "2026-04-13T05:40:52Z", "mode": "train", "global_step": 2923, "epoch": 0.2936212958312406, "loss": 0.0324, "grad_norm": 13.219432830810547, "learning_rate": 1.1454545454545457e-06, "num_tokens": 5460215.0, "completions/mean_length": 33.25, "completions/min_length": 32.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.25, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.8084553480148315, "rewards/meter/std": 0.3324153423309326, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9414901733398438, "rewards/repeat_soft/std": 0.03770037367939949, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7395789623260498, "rewards/total_composite/std": 0.15070529282093048, "reward": 0.7395789623260498, "reward_std": 0.15070529282093048, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08488535135984421, "sampling/sampling_logp_difference/max": 1.3350257873535156, "sampling/importance_sampling_ratio/min": 0.2631513774394989, "sampling/importance_sampling_ratio/mean": 1.0049757957458496, "sampling/importance_sampling_ratio/max": 1.800757646560669, "entropy": 0.3861354812979698, "clip_ratio/low_mean": 0.007464349502697587, "clip_ratio/low_min": 0.007464349502697587, "clip_ratio/high_mean": 0.07533283345401287, "clip_ratio/high_max": 0.07533283345401287, "clip_ratio/region_mean": 0.08279718295671046, "reward_total_mean": 0.7395789623260498, "reward_meter_mean": 0.8084553480148315, "reward_meter_std": 0.3324153423309326, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9414901733398438, "reward_repeat_soft_std": 0.03770037367939949, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7395789623260498, "reward_total_composite_std": 0.15070529282093048} {"timestamp_utc": "2026-04-13T05:41:00Z", "mode": "train", "global_step": 2924, "epoch": 0.2937217478653943, "loss": 0.0259, "grad_norm": 4.221101760864258, "learning_rate": 1.1424242424242425e-06, "num_tokens": 5462658.0, "completions/mean_length": 135.375, "completions/min_length": 117.0, "completions/max_length": 152.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 135.375, "completions/min_terminated_length": 117.0, "completions/max_terminated_length": 152.0, "rewards/meter/mean": 0.9891417026519775, "rewards/meter/std": 0.0037006221245974302, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6423592567443848, "rewards/repeat_soft/std": 0.0911720022559166, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.778974711894989, "rewards/total_composite/std": 0.022552331909537315, "reward": 0.778974711894989, "reward_std": 0.02255234867334366, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0544859804213047, "sampling/sampling_logp_difference/max": 2.6638600826263428, "sampling/importance_sampling_ratio/min": 0.06967873871326447, "sampling/importance_sampling_ratio/mean": 1.0002814531326294, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.21258785761892796, "clip_ratio/low_mean": 0.014664908405393362, "clip_ratio/low_min": 0.014664908405393362, "clip_ratio/high_mean": 0.0413556513376534, "clip_ratio/high_max": 0.0413556513376534, "clip_ratio/region_mean": 0.05602055974304676, "reward_total_mean": 0.778974711894989, "reward_meter_mean": 0.9891417026519775, "reward_meter_std": 0.0037006221245974302, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6423592567443848, "reward_repeat_soft_std": 0.0911720022559166, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.778974711894989, "reward_total_composite_std": 0.022552331909537315} {"timestamp_utc": "2026-04-13T05:41:09Z", "mode": "train", "global_step": 2925, "epoch": 0.29382219989954794, "loss": 0.0369, "grad_norm": 6.144402503967285, "learning_rate": 1.1393939393939395e-06, "num_tokens": 5465317.0, "completions/mean_length": 143.375, "completions/min_length": 135.0, "completions/max_length": 153.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 143.375, "completions/min_terminated_length": 135.0, "completions/max_terminated_length": 153.0, "rewards/meter/mean": 0.9788939952850342, "rewards/meter/std": 0.02178032323718071, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9469076991081238, "rewards/repeat_soft/std": 0.022036166861653328, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.8224430084228516, "rewards/total_composite/std": 0.023021021857857704, "reward": 0.8224430084228516, "reward_std": 0.023021018132567406, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08401458710432053, "sampling/sampling_logp_difference/max": 2.3584072589874268, "sampling/importance_sampling_ratio/min": 0.09457072615623474, "sampling/importance_sampling_ratio/mean": 1.0070735216140747, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.34605173394083977, "clip_ratio/low_mean": 0.056018145522102714, "clip_ratio/low_min": 0.056018145522102714, "clip_ratio/high_mean": 0.013586956076323986, "clip_ratio/high_max": 0.013586956076323986, "clip_ratio/region_mean": 0.0696051015984267, "reward_total_mean": 0.8224430084228516, "reward_meter_mean": 0.9788939952850342, "reward_meter_std": 0.02178032323718071, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9469076991081238, "reward_repeat_soft_std": 0.022036166861653328, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.8224430084228516, "reward_total_composite_std": 0.023021021857857704} {"timestamp_utc": "2026-04-13T05:41:21Z", "mode": "train", "global_step": 2926, "epoch": 0.29392265193370165, "loss": -0.0768, "grad_norm": 1.1605008840560913, "learning_rate": 1.1363636363636364e-06, "num_tokens": 5466707.0, "completions/mean_length": 149.75, "completions/min_length": 24.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.8414822816848755, "rewards/meter/std": 0.32381781935691833, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9514474868774414, "rewards/repeat_soft/std": 0.019620735198259354, "rewards/judge_quality/mean": 0.3462499976158142, "rewards/judge_quality/std": 0.18314221501350403, "rewards/total_composite/mean": 0.6188679337501526, "rewards/total_composite/std": 0.3819936215877533, "reward": 0.6188679337501526, "reward_std": 0.3819935917854309, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10187853127717972, "sampling/sampling_logp_difference/max": 1.1072232723236084, "sampling/importance_sampling_ratio/min": 0.3304753303527832, "sampling/importance_sampling_ratio/mean": 1.012873888015747, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3933093547821045, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.07434889674186707, "clip_ratio/high_max": 0.07434889674186707, "clip_ratio/region_mean": 0.07434889674186707, "reward_total_mean": 0.6188679337501526, "reward_meter_mean": 0.8414822816848755, "reward_meter_std": 0.32381781935691833, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9514474868774414, "reward_repeat_soft_std": 0.019620735198259354, "reward_judge_quality_mean": 0.3462499976158142, "reward_judge_quality_std": 0.18314221501350403, "reward_total_composite_mean": 0.6188679337501526, "reward_total_composite_std": 0.3819936215877533} {"timestamp_utc": "2026-04-13T05:41:29Z", "mode": "train", "global_step": 2927, "epoch": 0.29402310396785536, "loss": 0.0256, "grad_norm": 4.694357395172119, "learning_rate": 1.1333333333333334e-06, "num_tokens": 5469244.0, "completions/mean_length": 121.125, "completions/min_length": 111.0, "completions/max_length": 139.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 121.125, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.9630577564239502, "rewards/meter/std": 0.01979164592921734, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7121454477310181, "rewards/repeat_soft/std": 0.10208877176046371, "rewards/judge_quality/mean": 0.3137499690055847, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7412155866622925, "rewards/total_composite/std": 0.03498280048370361, "reward": 0.7412155866622925, "reward_std": 0.03498280048370361, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.048546187579631805, "sampling/sampling_logp_difference/max": 1.1341981887817383, "sampling/importance_sampling_ratio/min": 0.321679949760437, "sampling/importance_sampling_ratio/mean": 1.0083664655685425, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23542746156454086, "clip_ratio/low_mean": 0.02468717540614307, "clip_ratio/low_min": 0.02468717540614307, "clip_ratio/high_mean": 0.02144670858979225, "clip_ratio/high_max": 0.02144670858979225, "clip_ratio/region_mean": 0.04613388399593532, "reward_total_mean": 0.7412155866622925, "reward_meter_mean": 0.9630577564239502, "reward_meter_std": 0.01979164592921734, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7121454477310181, "reward_repeat_soft_std": 0.10208877176046371, "reward_judge_quality_mean": 0.3137499690055847, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7412155866622925, "reward_total_composite_std": 0.03498280048370361} {"timestamp_utc": "2026-04-13T05:41:36Z", "mode": "train", "global_step": 2928, "epoch": 0.29412355600200907, "loss": 0.0039, "grad_norm": 8.886374473571777, "learning_rate": 1.1303030303030305e-06, "num_tokens": 5470993.0, "completions/mean_length": 55.625, "completions/min_length": 45.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.625, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9940491914749146, "rewards/meter/std": 0.002522020135074854, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8405702114105225, "rewards/repeat_soft/std": 0.057523228228092194, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.8043791651725769, "rewards/total_composite/std": 0.018620038405060768, "reward": 0.8043791651725769, "reward_std": 0.01862003467977047, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07459378987550735, "sampling/sampling_logp_difference/max": 1.4305071830749512, "sampling/importance_sampling_ratio/min": 0.2391875833272934, "sampling/importance_sampling_ratio/mean": 0.9994475245475769, "sampling/importance_sampling_ratio/max": 1.6907386779785156, "entropy": 0.42699139937758446, "clip_ratio/low_mean": 0.019285301212221384, "clip_ratio/low_min": 0.019285301212221384, "clip_ratio/high_mean": 0.061226506251841784, "clip_ratio/high_max": 0.061226506251841784, "clip_ratio/region_mean": 0.08051180746406317, "reward_total_mean": 0.8043791651725769, "reward_meter_mean": 0.9940491914749146, "reward_meter_std": 0.002522020135074854, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8405702114105225, "reward_repeat_soft_std": 0.057523228228092194, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.8043791651725769, "reward_total_composite_std": 0.018620038405060768} {"timestamp_utc": "2026-04-13T05:41:43Z", "mode": "train", "global_step": 2929, "epoch": 0.2942240080361627, "loss": 0.0445, "grad_norm": 8.783370018005371, "learning_rate": 1.1272727272727275e-06, "num_tokens": 5472637.0, "completions/mean_length": 58.5, "completions/min_length": 52.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.5, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9899939894676208, "rewards/meter/std": 0.006039181258529425, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9798132181167603, "rewards/repeat_soft/std": 0.021011922508478165, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8194786310195923, "rewards/total_composite/std": 0.0029675979167222977, "reward": 0.8194786310195923, "reward_std": 0.002967605134472251, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08545612543821335, "sampling/sampling_logp_difference/max": 1.0556187629699707, "sampling/importance_sampling_ratio/min": 0.34797704219818115, "sampling/importance_sampling_ratio/mean": 1.0113763809204102, "sampling/importance_sampling_ratio/max": 1.961558222770691, "entropy": 0.4512406401336193, "clip_ratio/low_mean": 0.04092044220305979, "clip_ratio/low_min": 0.04092044220305979, "clip_ratio/high_mean": 0.042740863747894764, "clip_ratio/high_max": 0.042740863747894764, "clip_ratio/region_mean": 0.08366130595095456, "reward_total_mean": 0.8194786310195923, "reward_meter_mean": 0.9899939894676208, "reward_meter_std": 0.006039181258529425, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9798132181167603, "reward_repeat_soft_std": 0.021011922508478165, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8194786310195923, "reward_total_composite_std": 0.0029675979167222977} {"timestamp_utc": "2026-04-13T05:41:53Z", "mode": "train", "global_step": 2930, "epoch": 0.29432446007031643, "loss": 0.0168, "grad_norm": 4.033034801483154, "learning_rate": 1.1242424242424243e-06, "num_tokens": 5475785.0, "completions/mean_length": 200.5, "completions/min_length": 193.0, "completions/max_length": 211.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 200.5, "completions/min_terminated_length": 193.0, "completions/max_terminated_length": 211.0, "rewards/meter/mean": 0.9735754132270813, "rewards/meter/std": 0.05137484893202782, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7146693468093872, "rewards/repeat_soft/std": 0.08058232069015503, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7728258371353149, "rewards/total_composite/std": 0.030809609219431877, "reward": 0.7728258371353149, "reward_std": 0.030809614807367325, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06627904623746872, "sampling/sampling_logp_difference/max": 2.0404984951019287, "sampling/importance_sampling_ratio/min": 0.12996390461921692, "sampling/importance_sampling_ratio/mean": 1.0036003589630127, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.30421044677495956, "clip_ratio/low_mean": 0.024946993915364146, "clip_ratio/low_min": 0.024946993915364146, "clip_ratio/high_mean": 0.034955951385200024, "clip_ratio/high_max": 0.034955951385200024, "clip_ratio/region_mean": 0.05990294530056417, "reward_total_mean": 0.7728258371353149, "reward_meter_mean": 0.9735754132270813, "reward_meter_std": 0.05137484893202782, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7146693468093872, "reward_repeat_soft_std": 0.08058232069015503, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7728258371353149, "reward_total_composite_std": 0.030809609219431877} {"timestamp_utc": "2026-04-13T05:42:00Z", "mode": "train", "global_step": 2931, "epoch": 0.29442491210447014, "loss": -0.0192, "grad_norm": 16.279212951660156, "learning_rate": 1.1212121212121214e-06, "num_tokens": 5477389.0, "completions/mean_length": 39.5, "completions/min_length": 35.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.5, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.8629748225212097, "rewards/meter/std": 0.3023061752319336, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9755980968475342, "rewards/repeat_soft/std": 0.026606878265738487, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.764148473739624, "rewards/total_composite/std": 0.13629360496997833, "reward": 0.764148473739624, "reward_std": 0.13629359006881714, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10090178996324539, "sampling/sampling_logp_difference/max": 1.3634757995605469, "sampling/importance_sampling_ratio/min": 0.255770206451416, "sampling/importance_sampling_ratio/mean": 1.0104882717132568, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3703477270901203, "clip_ratio/low_mean": 0.020270269364118576, "clip_ratio/low_min": 0.020270269364118576, "clip_ratio/high_mean": 0.08389560016803443, "clip_ratio/high_max": 0.08389560016803443, "clip_ratio/region_mean": 0.10416586953215301, "reward_total_mean": 0.764148473739624, "reward_meter_mean": 0.8629748225212097, "reward_meter_std": 0.3023061752319336, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9755980968475342, "reward_repeat_soft_std": 0.026606878265738487, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.764148473739624, "reward_total_composite_std": 0.13629360496997833} {"timestamp_utc": "2026-04-13T05:42:06Z", "mode": "train", "global_step": 2932, "epoch": 0.2945253641386238, "loss": 0.0015, "grad_norm": 7.60252046585083, "learning_rate": 1.1181818181818182e-06, "num_tokens": 5479044.0, "completions/mean_length": 44.875, "completions/min_length": 40.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.875, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.9629055261611938, "rewards/meter/std": 0.0058615137822926044, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.906622588634491, "rewards/repeat_soft/std": 0.07792611420154572, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.8374698162078857, "rewards/total_composite/std": 0.07073991745710373, "reward": 0.8374698162078857, "reward_std": 0.07073991745710373, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07100556790828705, "sampling/sampling_logp_difference/max": 1.5411415100097656, "sampling/importance_sampling_ratio/min": 0.2141365259885788, "sampling/importance_sampling_ratio/mean": 0.9894653558731079, "sampling/importance_sampling_ratio/max": 1.415952444076538, "entropy": 0.25894713029265404, "clip_ratio/low_mean": 0.04719202849082649, "clip_ratio/low_min": 0.04719202849082649, "clip_ratio/high_mean": 0.01920289918780327, "clip_ratio/high_max": 0.01920289918780327, "clip_ratio/region_mean": 0.06639492767862976, "reward_total_mean": 0.8374698162078857, "reward_meter_mean": 0.9629055261611938, "reward_meter_std": 0.0058615137822926044, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.906622588634491, "reward_repeat_soft_std": 0.07792611420154572, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.8374698162078857, "reward_total_composite_std": 0.07073991745710373} {"timestamp_utc": "2026-04-13T05:42:12Z", "mode": "train", "global_step": 2933, "epoch": 0.2946258161727775, "loss": 0.0117, "grad_norm": 9.043015480041504, "learning_rate": 1.1151515151515153e-06, "num_tokens": 5480749.0, "completions/mean_length": 49.125, "completions/min_length": 43.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.125, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9848295450210571, "rewards/meter/std": 0.009312492795288563, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9348919987678528, "rewards/repeat_soft/std": 0.04177089035511017, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8149124979972839, "rewards/total_composite/std": 0.006970592774450779, "reward": 0.8149124979972839, "reward_std": 0.006970596499741077, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09390085935592651, "sampling/sampling_logp_difference/max": 1.4907578229904175, "sampling/importance_sampling_ratio/min": 0.22520193457603455, "sampling/importance_sampling_ratio/mean": 1.0187379121780396, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5335349291563034, "clip_ratio/low_mean": 0.05250049103051424, "clip_ratio/low_min": 0.05250049103051424, "clip_ratio/high_mean": 0.05683136032894254, "clip_ratio/high_max": 0.05683136032894254, "clip_ratio/region_mean": 0.10933185135945678, "reward_total_mean": 0.8149124979972839, "reward_meter_mean": 0.9848295450210571, "reward_meter_std": 0.009312492795288563, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9348919987678528, "reward_repeat_soft_std": 0.04177089035511017, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8149124979972839, "reward_total_composite_std": 0.006970592774450779} {"timestamp_utc": "2026-04-13T05:42:20Z", "mode": "train", "global_step": 2934, "epoch": 0.2947262682069312, "loss": -0.0486, "grad_norm": 4.364571571350098, "learning_rate": 1.112121212121212e-06, "num_tokens": 5483717.0, "completions/mean_length": 157.0, "completions/min_length": 125.0, "completions/max_length": 188.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 157.0, "completions/min_terminated_length": 125.0, "completions/max_terminated_length": 188.0, "rewards/meter/mean": 0.9938609600067139, "rewards/meter/std": 0.0031435855198651552, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9124120473861694, "rewards/repeat_soft/std": 0.025659197941422462, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.8043535947799683, "rewards/total_composite/std": 0.04889069125056267, "reward": 0.8043535947799683, "reward_std": 0.048890672624111176, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07864150404930115, "sampling/sampling_logp_difference/max": 4.453469753265381, "sampling/importance_sampling_ratio/min": 0.011638115160167217, "sampling/importance_sampling_ratio/mean": 1.0017657279968262, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37531669437885284, "clip_ratio/low_mean": 0.03885722439736128, "clip_ratio/low_min": 0.03885722439736128, "clip_ratio/high_mean": 0.04575853142887354, "clip_ratio/high_max": 0.04575853142887354, "clip_ratio/region_mean": 0.08461575582623482, "reward_total_mean": 0.8043535947799683, "reward_meter_mean": 0.9938609600067139, "reward_meter_std": 0.0031435855198651552, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9124120473861694, "reward_repeat_soft_std": 0.025659197941422462, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.8043535947799683, "reward_total_composite_std": 0.04889069125056267} {"timestamp_utc": "2026-04-13T05:42:32Z", "mode": "train", "global_step": 2935, "epoch": 0.29482672024108486, "loss": -0.0922, "grad_norm": 2.2870922088623047, "learning_rate": 1.1090909090909093e-06, "num_tokens": 5485161.0, "completions/mean_length": 88.5, "completions/min_length": 25.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 28.000001907348633, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.7757322788238525, "rewards/meter/std": 0.35237470269203186, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.8112499713897705, "rewards/judge_quality/std": 0.3075914680957794, "rewards/total_composite/mean": 0.8060475587844849, "rewards/total_composite/std": 0.3336544334888458, "reward": 0.8060475587844849, "reward_std": 0.33365437388420105, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14357760548591614, "sampling/sampling_logp_difference/max": 4.8914289474487305, "sampling/importance_sampling_ratio/min": 0.007510682567954063, "sampling/importance_sampling_ratio/mean": 1.0193911790847778, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5603581927716732, "clip_ratio/low_mean": 0.013392857275903225, "clip_ratio/low_min": 0.013392857275903225, "clip_ratio/high_mean": 0.09035980328917503, "clip_ratio/high_max": 0.09035980328917503, "clip_ratio/region_mean": 0.10375266056507826, "reward_total_mean": 0.8060475587844849, "reward_meter_mean": 0.7757322788238525, "reward_meter_std": 0.35237470269203186, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.8112499713897705, "reward_judge_quality_std": 0.3075914680957794, "reward_total_composite_mean": 0.8060475587844849, "reward_total_composite_std": 0.3336544334888458} {"timestamp_utc": "2026-04-13T05:42:38Z", "mode": "train", "global_step": 2936, "epoch": 0.29492717227523857, "loss": 0.0076, "grad_norm": 10.33514404296875, "learning_rate": 1.1060606060606062e-06, "num_tokens": 5486805.0, "completions/mean_length": 44.5, "completions/min_length": 37.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.5, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9668272733688354, "rewards/meter/std": 0.03511800989508629, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9395322799682617, "rewards/repeat_soft/std": 0.04986729472875595, "rewards/judge_quality/mean": 0.65625, "rewards/judge_quality/std": 0.23790381848812103, "rewards/total_composite/mean": 0.8759005069732666, "rewards/total_composite/std": 0.06949973851442337, "reward": 0.8759005069732666, "reward_std": 0.06949974596500397, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08353789895772934, "sampling/sampling_logp_difference/max": 2.2215704917907715, "sampling/importance_sampling_ratio/min": 0.10843867063522339, "sampling/importance_sampling_ratio/mean": 1.0168554782867432, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4141947776079178, "clip_ratio/low_mean": 0.02883657906204462, "clip_ratio/low_min": 0.02883657906204462, "clip_ratio/high_mean": 0.03767965408042073, "clip_ratio/high_max": 0.03767965408042073, "clip_ratio/region_mean": 0.06651623314246535, "reward_total_mean": 0.8759005069732666, "reward_meter_mean": 0.9668272733688354, "reward_meter_std": 0.03511800989508629, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9395322799682617, "reward_repeat_soft_std": 0.04986729472875595, "reward_judge_quality_mean": 0.65625, "reward_judge_quality_std": 0.23790381848812103, "reward_total_composite_mean": 0.8759005069732666, "reward_total_composite_std": 0.06949973851442337} {"timestamp_utc": "2026-04-13T05:42:44Z", "mode": "train", "global_step": 2937, "epoch": 0.2950276243093923, "loss": -0.0064, "grad_norm": 12.28377628326416, "learning_rate": 1.1030303030303032e-06, "num_tokens": 5488622.0, "completions/mean_length": 47.125, "completions/min_length": 42.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.125, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.9882407784461975, "rewards/meter/std": 0.006754573434591293, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.957305908203125, "rewards/repeat_soft/std": 0.032858312129974365, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.8100639581680298, "rewards/total_composite/std": 0.01726743020117283, "reward": 0.8100639581680298, "reward_std": 0.01726743020117283, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09242109954357147, "sampling/sampling_logp_difference/max": 1.3927998542785645, "sampling/importance_sampling_ratio/min": 0.2483789026737213, "sampling/importance_sampling_ratio/mean": 1.0174028873443604, "sampling/importance_sampling_ratio/max": 1.9641801118850708, "entropy": 0.5195322968065739, "clip_ratio/low_mean": 0.03569579962641001, "clip_ratio/low_min": 0.03569579962641001, "clip_ratio/high_mean": 0.07181885419413447, "clip_ratio/high_max": 0.07181885419413447, "clip_ratio/region_mean": 0.10751465382054448, "reward_total_mean": 0.8100639581680298, "reward_meter_mean": 0.9882407784461975, "reward_meter_std": 0.006754573434591293, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.957305908203125, "reward_repeat_soft_std": 0.032858312129974365, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.8100639581680298, "reward_total_composite_std": 0.01726743020117283} {"timestamp_utc": "2026-04-13T05:42:51Z", "mode": "train", "global_step": 2938, "epoch": 0.29512807634354593, "loss": 0.0101, "grad_norm": 7.258863925933838, "learning_rate": 1.1e-06, "num_tokens": 5491022.0, "completions/mean_length": 121.0, "completions/min_length": 113.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 121.0, "completions/min_terminated_length": 113.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.985653281211853, "rewards/meter/std": 0.012066461145877838, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8869892358779907, "rewards/repeat_soft/std": 0.04435469210147858, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7891179323196411, "rewards/total_composite/std": 0.025986477732658386, "reward": 0.7891179323196411, "reward_std": 0.025986477732658386, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07796671241521835, "sampling/sampling_logp_difference/max": 4.004810333251953, "sampling/importance_sampling_ratio/min": 0.018227744847536087, "sampling/importance_sampling_ratio/mean": 1.0055474042892456, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.34024618566036224, "clip_ratio/low_mean": 0.021113723050802946, "clip_ratio/low_min": 0.021113723050802946, "clip_ratio/high_mean": 0.05696999793872237, "clip_ratio/high_max": 0.05696999793872237, "clip_ratio/region_mean": 0.07808372098952532, "reward_total_mean": 0.7891179323196411, "reward_meter_mean": 0.985653281211853, "reward_meter_std": 0.012066461145877838, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8869892358779907, "reward_repeat_soft_std": 0.04435469210147858, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7891179323196411, "reward_total_composite_std": 0.025986477732658386} {"timestamp_utc": "2026-04-13T05:42:58Z", "mode": "train", "global_step": 2939, "epoch": 0.29522852837769964, "loss": 0.034, "grad_norm": 7.366160869598389, "learning_rate": 1.096969696969697e-06, "num_tokens": 5492882.0, "completions/mean_length": 46.5, "completions/min_length": 42.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.5, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9614681005477905, "rewards/meter/std": 0.0043991245329380035, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.84139484167099, "rewards/repeat_soft/std": 0.1426503211259842, "rewards/judge_quality/mean": 0.3762499988079071, "rewards/judge_quality/std": 0.11287634819746017, "rewards/total_composite/mean": 0.7796751260757446, "rewards/total_composite/std": 0.04347499832510948, "reward": 0.7796751260757446, "reward_std": 0.04347499459981918, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06793443858623505, "sampling/sampling_logp_difference/max": 0.9932646751403809, "sampling/importance_sampling_ratio/min": 0.37036558985710144, "sampling/importance_sampling_ratio/mean": 0.9953596591949463, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.28685317747294903, "clip_ratio/low_mean": 0.024003622820600867, "clip_ratio/low_min": 0.024003622820600867, "clip_ratio/high_mean": 0.06262939888983965, "clip_ratio/high_max": 0.06262939888983965, "clip_ratio/region_mean": 0.08663302171044052, "reward_total_mean": 0.7796751260757446, "reward_meter_mean": 0.9614681005477905, "reward_meter_std": 0.0043991245329380035, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.84139484167099, "reward_repeat_soft_std": 0.1426503211259842, "reward_judge_quality_mean": 0.3762499988079071, "reward_judge_quality_std": 0.11287634819746017, "reward_total_composite_mean": 0.7796751260757446, "reward_total_composite_std": 0.04347499832510948} {"timestamp_utc": "2026-04-13T05:43:04Z", "mode": "train", "global_step": 2940, "epoch": 0.29532898041185335, "loss": 0.0076, "grad_norm": 5.742631435394287, "learning_rate": 1.093939393939394e-06, "num_tokens": 5494633.0, "completions/mean_length": 57.875, "completions/min_length": 51.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.875, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9907137155532837, "rewards/meter/std": 0.003613231470808387, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.953048586845398, "rewards/repeat_soft/std": 0.04613584280014038, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.8295010328292847, "rewards/total_composite/std": 0.030833808705210686, "reward": 0.8295010328292847, "reward_std": 0.030833804979920387, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07282204926013947, "sampling/sampling_logp_difference/max": 1.0664259195327759, "sampling/importance_sampling_ratio/min": 0.3442366421222687, "sampling/importance_sampling_ratio/mean": 1.0155941247940063, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3522844649851322, "clip_ratio/low_mean": 0.05233656708151102, "clip_ratio/low_min": 0.05233656708151102, "clip_ratio/high_mean": 0.008340523578226566, "clip_ratio/high_max": 0.008340523578226566, "clip_ratio/region_mean": 0.06067709065973759, "reward_total_mean": 0.8295010328292847, "reward_meter_mean": 0.9907137155532837, "reward_meter_std": 0.003613231470808387, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.953048586845398, "reward_repeat_soft_std": 0.04613584280014038, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.8295010328292847, "reward_total_composite_std": 0.030833808705210686} {"timestamp_utc": "2026-04-13T05:43:12Z", "mode": "train", "global_step": 2941, "epoch": 0.29542943244600706, "loss": 0.0437, "grad_norm": 5.513673782348633, "learning_rate": 1.090909090909091e-06, "num_tokens": 5497274.0, "completions/mean_length": 151.125, "completions/min_length": 140.0, "completions/max_length": 167.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 151.125, "completions/min_terminated_length": 140.0, "completions/max_terminated_length": 167.0, "rewards/meter/mean": 0.9644657373428345, "rewards/meter/std": 0.08492974936962128, "rewards/count_adherence/mean": 0.8500000238418579, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.896440863609314, "rewards/repeat_soft/std": 0.03518875688314438, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7771536707878113, "rewards/total_composite/std": 0.030082618817687035, "reward": 0.7771536707878113, "reward_std": 0.030082600191235542, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07397760450839996, "sampling/sampling_logp_difference/max": 2.191990613937378, "sampling/importance_sampling_ratio/min": 0.11169418692588806, "sampling/importance_sampling_ratio/mean": 1.0084850788116455, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.41929732263088226, "clip_ratio/low_mean": 0.005239521153271198, "clip_ratio/low_min": 0.005239521153271198, "clip_ratio/high_mean": 0.058696275344118476, "clip_ratio/high_max": 0.058696275344118476, "clip_ratio/region_mean": 0.06393579649738967, "reward_total_mean": 0.7771536707878113, "reward_meter_mean": 0.9644657373428345, "reward_meter_std": 0.08492974936962128, "reward_count_adherence_mean": 0.8500000238418579, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.896440863609314, "reward_repeat_soft_std": 0.03518875688314438, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7771536707878113, "reward_total_composite_std": 0.030082618817687035} {"timestamp_utc": "2026-04-13T05:43:20Z", "mode": "train", "global_step": 2942, "epoch": 0.2955298844801607, "loss": 0.0072, "grad_norm": 4.3737382888793945, "learning_rate": 1.087878787878788e-06, "num_tokens": 5500168.0, "completions/mean_length": 175.75, "completions/min_length": 148.0, "completions/max_length": 194.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 175.75, "completions/min_terminated_length": 148.0, "completions/max_terminated_length": 194.0, "rewards/meter/mean": 0.9927089214324951, "rewards/meter/std": 0.005827242974191904, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8686295747756958, "rewards/repeat_soft/std": 0.07145687937736511, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7957069873809814, "rewards/total_composite/std": 0.028572728857398033, "reward": 0.7957069873809814, "reward_std": 0.02857271023094654, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07100100070238113, "sampling/sampling_logp_difference/max": 1.8320338726043701, "sampling/importance_sampling_ratio/min": 0.16008764505386353, "sampling/importance_sampling_ratio/mean": 1.0032740831375122, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3428202345967293, "clip_ratio/low_mean": 0.013006197987124324, "clip_ratio/low_min": 0.013006197987124324, "clip_ratio/high_mean": 0.054768905974924564, "clip_ratio/high_max": 0.054768905974924564, "clip_ratio/region_mean": 0.06777510396204889, "reward_total_mean": 0.7957069873809814, "reward_meter_mean": 0.9927089214324951, "reward_meter_std": 0.005827242974191904, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8686295747756958, "reward_repeat_soft_std": 0.07145687937736511, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7957069873809814, "reward_total_composite_std": 0.028572728857398033} {"timestamp_utc": "2026-04-13T05:43:28Z", "mode": "train", "global_step": 2943, "epoch": 0.2956303365143144, "loss": -0.0049, "grad_norm": 5.902942657470703, "learning_rate": 1.084848484848485e-06, "num_tokens": 5502639.0, "completions/mean_length": 120.875, "completions/min_length": 115.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.875, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9473534822463989, "rewards/meter/std": 0.07906533777713776, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.953620433807373, "rewards/repeat_soft/std": 0.030171658843755722, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.8164211511611938, "rewards/total_composite/std": 0.06837371736764908, "reward": 0.8164211511611938, "reward_std": 0.06837372481822968, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09433455020189285, "sampling/sampling_logp_difference/max": 3.161266326904297, "sampling/importance_sampling_ratio/min": 0.08902116119861603, "sampling/importance_sampling_ratio/mean": 1.0008381605148315, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3573102820664644, "clip_ratio/low_mean": 0.04279460525140166, "clip_ratio/low_min": 0.04279460525140166, "clip_ratio/high_mean": 0.03543545678257942, "clip_ratio/high_max": 0.03543545678257942, "clip_ratio/region_mean": 0.07823006203398108, "reward_total_mean": 0.8164211511611938, "reward_meter_mean": 0.9473534822463989, "reward_meter_std": 0.07906533777713776, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.953620433807373, "reward_repeat_soft_std": 0.030171658843755722, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.8164211511611938, "reward_total_composite_std": 0.06837371736764908} {"timestamp_utc": "2026-04-13T05:43:35Z", "mode": "train", "global_step": 2944, "epoch": 0.2957307885484681, "loss": 0.013, "grad_norm": 3.093876600265503, "learning_rate": 1.081818181818182e-06, "num_tokens": 5505182.0, "completions/mean_length": 131.875, "completions/min_length": 120.0, "completions/max_length": 139.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.875, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.9888243675231934, "rewards/meter/std": 0.00878660287708044, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7144302129745483, "rewards/repeat_soft/std": 0.11266880482435226, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.8111640214920044, "rewards/total_composite/std": 0.05168123170733452, "reward": 0.8111640214920044, "reward_std": 0.05168122425675392, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06258794665336609, "sampling/sampling_logp_difference/max": 2.9438884258270264, "sampling/importance_sampling_ratio/min": 0.052660562098026276, "sampling/importance_sampling_ratio/mean": 1.00316321849823, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2543198484927416, "clip_ratio/low_mean": 0.05019176518544555, "clip_ratio/low_min": 0.05019176518544555, "clip_ratio/high_mean": 0.008395522832870483, "clip_ratio/high_max": 0.008395522832870483, "clip_ratio/region_mean": 0.05858728801831603, "reward_total_mean": 0.8111640214920044, "reward_meter_mean": 0.9888243675231934, "reward_meter_std": 0.00878660287708044, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7144302129745483, "reward_repeat_soft_std": 0.11266880482435226, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.8111640214920044, "reward_total_composite_std": 0.05168123170733452} {"timestamp_utc": "2026-04-13T05:43:42Z", "mode": "train", "global_step": 2945, "epoch": 0.2958312405826218, "loss": 0.0419, "grad_norm": 6.442337512969971, "learning_rate": 1.078787878787879e-06, "num_tokens": 5507447.0, "completions/mean_length": 110.125, "completions/min_length": 101.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.125, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.9583063125610352, "rewards/meter/std": 0.05984104424715042, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9463690519332886, "rewards/repeat_soft/std": 0.02061876468360424, "rewards/judge_quality/mean": 0.5575000047683716, "rewards/judge_quality/std": 0.19955310225486755, "rewards/total_composite/mean": 0.8431247472763062, "rewards/total_composite/std": 0.07306424528360367, "reward": 0.8431247472763062, "reward_std": 0.07306423783302307, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08834285289049149, "sampling/sampling_logp_difference/max": 2.5757248401641846, "sampling/importance_sampling_ratio/min": 0.07609864324331284, "sampling/importance_sampling_ratio/mean": 1.000030755996704, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45106708630919456, "clip_ratio/low_mean": 0.05196088319644332, "clip_ratio/low_min": 0.05196088319644332, "clip_ratio/high_mean": 0.034836350940167904, "clip_ratio/high_max": 0.034836350940167904, "clip_ratio/region_mean": 0.08679723413661122, "reward_total_mean": 0.8431247472763062, "reward_meter_mean": 0.9583063125610352, "reward_meter_std": 0.05984104424715042, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9463690519332886, "reward_repeat_soft_std": 0.02061876468360424, "reward_judge_quality_mean": 0.5575000047683716, "reward_judge_quality_std": 0.19955310225486755, "reward_total_composite_mean": 0.8431247472763062, "reward_total_composite_std": 0.07306424528360367} {"timestamp_utc": "2026-04-13T05:43:48Z", "mode": "train", "global_step": 2946, "epoch": 0.2959316926167755, "loss": 0.0251, "grad_norm": 10.959081649780273, "learning_rate": 1.0757575757575758e-06, "num_tokens": 5508806.0, "completions/mean_length": 25.875, "completions/min_length": 25.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.875, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.9745393991470337, "rewards/meter/std": 0.01460348442196846, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9327749013900757, "rewards/repeat_soft/std": 0.044481489807367325, "rewards/judge_quality/mean": 0.44999998807907104, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8168202638626099, "rewards/total_composite/std": 0.0076598431915044785, "reward": 0.8168202638626099, "reward_std": 0.007659862283617258, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06724052876234055, "sampling/sampling_logp_difference/max": 1.2965965270996094, "sampling/importance_sampling_ratio/min": 0.27346092462539673, "sampling/importance_sampling_ratio/mean": 1.0008881092071533, "sampling/importance_sampling_ratio/max": 1.8535683155059814, "entropy": 0.35058723017573357, "clip_ratio/low_mean": 0.033119658939540386, "clip_ratio/low_min": 0.033119658939540386, "clip_ratio/high_mean": 0.057955839205533266, "clip_ratio/high_max": 0.057955839205533266, "clip_ratio/region_mean": 0.09107549814507365, "reward_total_mean": 0.8168202638626099, "reward_meter_mean": 0.9745393991470337, "reward_meter_std": 0.01460348442196846, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9327749013900757, "reward_repeat_soft_std": 0.044481489807367325, "reward_judge_quality_mean": 0.44999998807907104, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8168202638626099, "reward_total_composite_std": 0.0076598431915044785} {"timestamp_utc": "2026-04-13T05:43:56Z", "mode": "train", "global_step": 2947, "epoch": 0.2960321446509292, "loss": 0.0111, "grad_norm": 8.872123718261719, "learning_rate": 1.0727272727272728e-06, "num_tokens": 5510779.0, "completions/mean_length": 85.625, "completions/min_length": 84.0, "completions/max_length": 89.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.625, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 89.0, "rewards/meter/mean": 0.9319913387298584, "rewards/meter/std": 0.1281941682100296, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9764307737350464, "rewards/repeat_soft/std": 0.013666026294231415, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.7765392065048218, "rewards/total_composite/std": 0.05745121091604233, "reward": 0.7765392065048218, "reward_std": 0.057451214641332626, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09858594089746475, "sampling/sampling_logp_difference/max": 1.5571794509887695, "sampling/importance_sampling_ratio/min": 0.21072959899902344, "sampling/importance_sampling_ratio/mean": 1.0087834596633911, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4809045009315014, "clip_ratio/low_mean": 0.03398109320551157, "clip_ratio/low_min": 0.03398109320551157, "clip_ratio/high_mean": 0.06656268099322915, "clip_ratio/high_max": 0.06656268099322915, "clip_ratio/region_mean": 0.10054377419874072, "reward_total_mean": 0.7765392065048218, "reward_meter_mean": 0.9319913387298584, "reward_meter_std": 0.1281941682100296, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9764307737350464, "reward_repeat_soft_std": 0.013666026294231415, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.7765392065048218, "reward_total_composite_std": 0.05745121091604233} {"timestamp_utc": "2026-04-13T05:44:07Z", "mode": "train", "global_step": 2948, "epoch": 0.29613259668508285, "loss": -0.1318, "grad_norm": 1.524308443069458, "learning_rate": 1.0696969696969696e-06, "num_tokens": 5512419.0, "completions/mean_length": 104.0, "completions/min_length": 44.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 45.71428680419922, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9323098659515381, "rewards/meter/std": 0.07659494876861572, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9949036836624146, "rewards/repeat_soft/std": 0.01107998937368393, "rewards/judge_quality/mean": 0.39625000953674316, "rewards/judge_quality/std": 0.14029940962791443, "rewards/total_composite/mean": 0.7012737393379211, "rewards/total_composite/std": 0.28539031744003296, "reward": 0.7012737393379211, "reward_std": 0.28539028763771057, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07467394322156906, "sampling/sampling_logp_difference/max": 1.1052258014678955, "sampling/importance_sampling_ratio/min": 0.3311360776424408, "sampling/importance_sampling_ratio/mean": 1.008422613143921, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2757852338254452, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06282541062682867, "clip_ratio/high_max": 0.06282541062682867, "clip_ratio/region_mean": 0.06282541062682867, "reward_total_mean": 0.7012737393379211, "reward_meter_mean": 0.9323098659515381, "reward_meter_std": 0.07659494876861572, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9949036836624146, "reward_repeat_soft_std": 0.01107998937368393, "reward_judge_quality_mean": 0.39625000953674316, "reward_judge_quality_std": 0.14029940962791443, "reward_total_composite_mean": 0.7012737393379211, "reward_total_composite_std": 0.28539031744003296} {"timestamp_utc": "2026-04-13T05:44:15Z", "mode": "train", "global_step": 2949, "epoch": 0.29623304871923656, "loss": 0.0082, "grad_norm": 4.834122657775879, "learning_rate": 1.066666666666667e-06, "num_tokens": 5515069.0, "completions/mean_length": 155.25, "completions/min_length": 150.0, "completions/max_length": 165.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 155.25, "completions/min_terminated_length": 150.0, "completions/max_terminated_length": 165.0, "rewards/meter/mean": 0.9557830095291138, "rewards/meter/std": 0.07762451469898224, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8535239100456238, "rewards/repeat_soft/std": 0.028421960771083832, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7850797176361084, "rewards/total_composite/std": 0.03656072914600372, "reward": 0.7850797176361084, "reward_std": 0.03656071797013283, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06816783547401428, "sampling/sampling_logp_difference/max": 1.5066509246826172, "sampling/importance_sampling_ratio/min": 0.22165106236934662, "sampling/importance_sampling_ratio/mean": 0.9985455870628357, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2693379931151867, "clip_ratio/low_mean": 0.009015151765197515, "clip_ratio/low_min": 0.009015151765197515, "clip_ratio/high_mean": 0.04012687038630247, "clip_ratio/high_max": 0.04012687038630247, "clip_ratio/region_mean": 0.04914202215149999, "reward_total_mean": 0.7850797176361084, "reward_meter_mean": 0.9557830095291138, "reward_meter_std": 0.07762451469898224, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8535239100456238, "reward_repeat_soft_std": 0.028421960771083832, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7850797176361084, "reward_total_composite_std": 0.03656072914600372} {"timestamp_utc": "2026-04-13T05:44:23Z", "mode": "train", "global_step": 2950, "epoch": 0.29633350075339027, "loss": 0.0147, "grad_norm": 7.588573932647705, "learning_rate": 1.0636363636363637e-06, "num_tokens": 5517546.0, "completions/mean_length": 137.625, "completions/min_length": 134.0, "completions/max_length": 147.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 137.625, "completions/min_terminated_length": 134.0, "completions/max_terminated_length": 147.0, "rewards/meter/mean": 0.9891536235809326, "rewards/meter/std": 0.003564384300261736, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7054414749145508, "rewards/repeat_soft/std": 0.07576955854892731, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.8291632533073425, "rewards/total_composite/std": 0.06879906356334686, "reward": 0.8291632533073425, "reward_std": 0.06879905611276627, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05855877697467804, "sampling/sampling_logp_difference/max": 1.4136905670166016, "sampling/importance_sampling_ratio/min": 0.24324393272399902, "sampling/importance_sampling_ratio/mean": 1.0037901401519775, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.27575729228556156, "clip_ratio/low_mean": 0.029566138400696218, "clip_ratio/low_min": 0.029566138400696218, "clip_ratio/high_mean": 0.007347021019086242, "clip_ratio/high_max": 0.007347021019086242, "clip_ratio/region_mean": 0.03691315941978246, "reward_total_mean": 0.8291632533073425, "reward_meter_mean": 0.9891536235809326, "reward_meter_std": 0.003564384300261736, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7054414749145508, "reward_repeat_soft_std": 0.07576955854892731, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.8291632533073425, "reward_total_composite_std": 0.06879906356334686} {"timestamp_utc": "2026-04-13T05:45:17Z", "mode": "eval", "global_step": 2950, "epoch": 0.29633350075339027, "eval_loss": NaN, "eval_runtime": 53.2256, "eval_samples_per_second": 1.503, "eval_steps_per_second": 0.188, "eval_num_tokens": 5517546.0, "eval_completions/mean_length": 101.925, "eval_completions/min_length": 42.8, "eval_completions/max_length": 226.9, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 91.61607208251954, "eval_completions/min_terminated_length": 42.8, "eval_completions/max_terminated_length": 161.1, "eval_rewards/meter/mean": 0.8922822415828705, "eval_rewards/meter/std": 0.19876350201666354, "eval_rewards/count_adherence/mean": 0.9706250011920929, "eval_rewards/count_adherence/std": 0.07825144529342651, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.07071067690849304, "eval_rewards/repeat_soft/mean": 0.9117227435112, "eval_rewards/repeat_soft/std": 0.08617102336138487, "eval_rewards/judge_quality/mean": 0.42399999499320984, "eval_rewards/judge_quality/std": 0.12000444103032351, "eval_rewards/total_composite/mean": 0.7567905485630035, "eval_rewards/total_composite/std": 0.13718529418110847, "eval_reward": 0.7567905485630035, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.041724132001399995, "eval_sampling/sampling_logp_difference/max": 1.014983057975769, "eval_sampling/importance_sampling_ratio/min": 0.36835823506116866, "eval_sampling/importance_sampling_ratio/mean": 1.0057571291923524, "eval_sampling/importance_sampling_ratio/max": 1.3531436800956727, "eval_entropy": 0.40603837072849275, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7567905485630035, "eval_reward_meter_mean": 0.8922822415828705, "eval_reward_meter_std": 0.19876350201666354, "eval_reward_count_adherence_mean": 0.9706250011920929, "eval_reward_count_adherence_std": 0.07825144529342651, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.07071067690849304, "eval_reward_repeat_soft_mean": 0.9117227435112, "eval_reward_repeat_soft_std": 0.08617102336138487, "eval_reward_judge_quality_mean": 0.42399999499320984, "eval_reward_judge_quality_std": 0.12000444103032351, "eval_reward_total_composite_mean": 0.7567905485630035, "eval_reward_total_composite_std": 0.13718529418110847} {"timestamp_utc": "2026-04-13T05:45:25Z", "mode": "train", "global_step": 2951, "epoch": 0.296433952787544, "loss": -0.0298, "grad_norm": 11.085311889648438, "learning_rate": 1.0606060606060608e-06, "num_tokens": 5519285.0, "completions/mean_length": 30.375, "completions/min_length": 26.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.375, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.6191094517707825, "rewards/meter/std": 0.4012030363082886, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9621384143829346, "rewards/repeat_soft/std": 0.0010226722806692123, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.647813081741333, "rewards/total_composite/std": 0.17566363513469696, "reward": 0.647813081741333, "reward_std": 0.17566363513469696, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11546177417039871, "sampling/sampling_logp_difference/max": 1.4489383697509766, "sampling/importance_sampling_ratio/min": 0.2348194420337677, "sampling/importance_sampling_ratio/mean": 0.9988914132118225, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5788577310740948, "clip_ratio/low_mean": 0.05160648608580232, "clip_ratio/low_min": 0.05160648608580232, "clip_ratio/high_mean": 0.0704445093870163, "clip_ratio/high_max": 0.0704445093870163, "clip_ratio/region_mean": 0.12205099547281861, "reward_total_mean": 0.647813081741333, "reward_meter_mean": 0.6191094517707825, "reward_meter_std": 0.4012030363082886, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9621384143829346, "reward_repeat_soft_std": 0.0010226722806692123, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.647813081741333, "reward_total_composite_std": 0.17566363513469696} {"timestamp_utc": "2026-04-13T05:45:33Z", "mode": "train", "global_step": 2952, "epoch": 0.29653440482169763, "loss": 0.0378, "grad_norm": 6.177953243255615, "learning_rate": 1.0575757575757576e-06, "num_tokens": 5521527.0, "completions/mean_length": 110.25, "completions/min_length": 100.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.25, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.9885357618331909, "rewards/meter/std": 0.008579098619520664, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9379022121429443, "rewards/repeat_soft/std": 0.042817309498786926, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.82588130235672, "rewards/total_composite/std": 0.034026846289634705, "reward": 0.82588130235672, "reward_std": 0.03402683511376381, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08927991986274719, "sampling/sampling_logp_difference/max": 1.6721701622009277, "sampling/importance_sampling_ratio/min": 0.18783897161483765, "sampling/importance_sampling_ratio/mean": 0.9992867112159729, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4554686062037945, "clip_ratio/low_mean": 0.0665777400135994, "clip_ratio/low_min": 0.0665777400135994, "clip_ratio/high_mean": 0.0062500000931322575, "clip_ratio/high_max": 0.0062500000931322575, "clip_ratio/region_mean": 0.07282774010673165, "reward_total_mean": 0.82588130235672, "reward_meter_mean": 0.9885357618331909, "reward_meter_std": 0.008579098619520664, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9379022121429443, "reward_repeat_soft_std": 0.042817309498786926, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.82588130235672, "reward_total_composite_std": 0.034026846289634705} {"timestamp_utc": "2026-04-13T05:45:40Z", "mode": "train", "global_step": 2953, "epoch": 0.29663485685585134, "loss": 0.0139, "grad_norm": 12.852466583251953, "learning_rate": 1.0545454545454547e-06, "num_tokens": 5523193.0, "completions/mean_length": 50.25, "completions/min_length": 47.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.25, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9809861183166504, "rewards/meter/std": 0.014829510822892189, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9559388756752014, "rewards/repeat_soft/std": 0.037503089755773544, "rewards/judge_quality/mean": 0.4312499761581421, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8164125680923462, "rewards/total_composite/std": 0.010876827873289585, "reward": 0.8164125680923462, "reward_std": 0.010876826010644436, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11179962009191513, "sampling/sampling_logp_difference/max": 3.481773853302002, "sampling/importance_sampling_ratio/min": 0.030752811580896378, "sampling/importance_sampling_ratio/mean": 0.9959455132484436, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6293473355472088, "clip_ratio/low_mean": 0.027254901826381683, "clip_ratio/low_min": 0.027254901826381683, "clip_ratio/high_mean": 0.06279172701761127, "clip_ratio/high_max": 0.06279172701761127, "clip_ratio/region_mean": 0.09004662884399295, "reward_total_mean": 0.8164125680923462, "reward_meter_mean": 0.9809861183166504, "reward_meter_std": 0.014829510822892189, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9559388756752014, "reward_repeat_soft_std": 0.037503089755773544, "reward_judge_quality_mean": 0.4312499761581421, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8164125680923462, "reward_total_composite_std": 0.010876827873289585} {"timestamp_utc": "2026-04-13T05:45:46Z", "mode": "train", "global_step": 2954, "epoch": 0.29673530889000505, "loss": 0.0364, "grad_norm": 8.982366561889648, "learning_rate": 1.0515151515151515e-06, "num_tokens": 5525078.0, "completions/mean_length": 56.625, "completions/min_length": 48.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.625, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9584395885467529, "rewards/meter/std": 0.07798094302415848, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9324177503585815, "rewards/repeat_soft/std": 0.0615503266453743, "rewards/judge_quality/mean": 0.36374998092651367, "rewards/judge_quality/std": 0.09500939399003983, "rewards/total_composite/mean": 0.7836645841598511, "rewards/total_composite/std": 0.03669416159391403, "reward": 0.7836645841598511, "reward_std": 0.03669416531920433, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07522963732481003, "sampling/sampling_logp_difference/max": 1.468519687652588, "sampling/importance_sampling_ratio/min": 0.23026609420776367, "sampling/importance_sampling_ratio/mean": 1.0036317110061646, "sampling/importance_sampling_ratio/max": 1.7769957780838013, "entropy": 0.3637390583753586, "clip_ratio/low_mean": 0.026361702010035515, "clip_ratio/low_min": 0.026361702010035515, "clip_ratio/high_mean": 0.04281426616944373, "clip_ratio/high_max": 0.04281426616944373, "clip_ratio/region_mean": 0.06917596817947924, "reward_total_mean": 0.7836645841598511, "reward_meter_mean": 0.9584395885467529, "reward_meter_std": 0.07798094302415848, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9324177503585815, "reward_repeat_soft_std": 0.0615503266453743, "reward_judge_quality_mean": 0.36374998092651367, "reward_judge_quality_std": 0.09500939399003983, "reward_total_composite_mean": 0.7836645841598511, "reward_total_composite_std": 0.03669416159391403} {"timestamp_utc": "2026-04-13T05:45:54Z", "mode": "train", "global_step": 2955, "epoch": 0.2968357609241587, "loss": 0.0142, "grad_norm": 5.481365203857422, "learning_rate": 1.0484848484848485e-06, "num_tokens": 5527430.0, "completions/mean_length": 120.0, "completions/min_length": 111.0, "completions/max_length": 127.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.0, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.988532304763794, "rewards/meter/std": 0.004358960781246424, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8088774681091309, "rewards/repeat_soft/std": 0.06145404651761055, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7889772653579712, "rewards/total_composite/std": 0.021527474746108055, "reward": 0.7889772653579712, "reward_std": 0.021527469158172607, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08076277375221252, "sampling/sampling_logp_difference/max": 2.1997499465942383, "sampling/importance_sampling_ratio/min": 0.1108308658003807, "sampling/importance_sampling_ratio/mean": 1.0062072277069092, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.42405957728624344, "clip_ratio/low_mean": 0.02512778202071786, "clip_ratio/low_min": 0.02512778202071786, "clip_ratio/high_mean": 0.059633033350110054, "clip_ratio/high_max": 0.059633033350110054, "clip_ratio/region_mean": 0.08476081537082791, "reward_total_mean": 0.7889772653579712, "reward_meter_mean": 0.988532304763794, "reward_meter_std": 0.004358960781246424, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8088774681091309, "reward_repeat_soft_std": 0.06145404651761055, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7889772653579712, "reward_total_composite_std": 0.021527474746108055} {"timestamp_utc": "2026-04-13T05:46:01Z", "mode": "train", "global_step": 2956, "epoch": 0.2969362129583124, "loss": 0.0137, "grad_norm": 6.878072261810303, "learning_rate": 1.0454545454545456e-06, "num_tokens": 5529536.0, "completions/mean_length": 72.25, "completions/min_length": 70.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.25, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9867876768112183, "rewards/meter/std": 0.0040461719036102295, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8175968527793884, "rewards/repeat_soft/std": 0.0957336500287056, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.1940544992685318, "rewards/total_composite/mean": 0.8153141736984253, "rewards/total_composite/std": 0.05522056668996811, "reward": 0.8153141736984253, "reward_std": 0.05522055923938751, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06433664262294769, "sampling/sampling_logp_difference/max": 2.2354495525360107, "sampling/importance_sampling_ratio/min": 0.10694403946399689, "sampling/importance_sampling_ratio/mean": 1.0058480501174927, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3348645493388176, "clip_ratio/low_mean": 0.058395927771925926, "clip_ratio/low_min": 0.058395927771925926, "clip_ratio/high_mean": 0.012206458020955324, "clip_ratio/high_max": 0.012206458020955324, "clip_ratio/region_mean": 0.07060238579288125, "reward_total_mean": 0.8153141736984253, "reward_meter_mean": 0.9867876768112183, "reward_meter_std": 0.0040461719036102295, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8175968527793884, "reward_repeat_soft_std": 0.0957336500287056, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.1940544992685318, "reward_total_composite_mean": 0.8153141736984253, "reward_total_composite_std": 0.05522056668996811} {"timestamp_utc": "2026-04-13T05:46:09Z", "mode": "train", "global_step": 2957, "epoch": 0.2970366649924661, "loss": 0.0035, "grad_norm": 4.382372856140137, "learning_rate": 1.0424242424242426e-06, "num_tokens": 5532772.0, "completions/mean_length": 200.5, "completions/min_length": 189.0, "completions/max_length": 215.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 200.5, "completions/min_terminated_length": 189.0, "completions/max_terminated_length": 215.0, "rewards/meter/mean": 0.9884700775146484, "rewards/meter/std": 0.002254422986879945, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7044712901115417, "rewards/repeat_soft/std": 0.10547629743814468, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.8287586569786072, "rewards/total_composite/std": 0.06888226419687271, "reward": 0.8287586569786072, "reward_std": 0.06888225674629211, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06093139946460724, "sampling/sampling_logp_difference/max": 2.67299485206604, "sampling/importance_sampling_ratio/min": 0.06904513388872147, "sampling/importance_sampling_ratio/mean": 0.997728168964386, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.24643555469810963, "clip_ratio/low_mean": 0.03784013958647847, "clip_ratio/low_min": 0.03784013958647847, "clip_ratio/high_mean": 0.014374019112437963, "clip_ratio/high_max": 0.014374019112437963, "clip_ratio/region_mean": 0.052214158698916435, "reward_total_mean": 0.8287586569786072, "reward_meter_mean": 0.9884700775146484, "reward_meter_std": 0.002254422986879945, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7044712901115417, "reward_repeat_soft_std": 0.10547629743814468, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.8287586569786072, "reward_total_composite_std": 0.06888226419687271} {"timestamp_utc": "2026-04-13T05:46:17Z", "mode": "train", "global_step": 2958, "epoch": 0.29713711702661977, "loss": -0.0182, "grad_norm": 7.0611042976379395, "learning_rate": 1.0393939393939394e-06, "num_tokens": 5535062.0, "completions/mean_length": 110.25, "completions/min_length": 90.0, "completions/max_length": 127.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.25, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.989596426486969, "rewards/meter/std": 0.007186277769505978, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7901089191436768, "rewards/repeat_soft/std": 0.058200206607580185, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.7958292365074158, "rewards/total_composite/std": 0.04314915090799332, "reward": 0.7958292365074158, "reward_std": 0.043149158358573914, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07020079344511032, "sampling/sampling_logp_difference/max": 1.2743196487426758, "sampling/importance_sampling_ratio/min": 0.2796211242675781, "sampling/importance_sampling_ratio/mean": 1.0015815496444702, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4157298170030117, "clip_ratio/low_mean": 0.01531346095725894, "clip_ratio/low_min": 0.01531346095725894, "clip_ratio/high_mean": 0.055105622159317136, "clip_ratio/high_max": 0.055105622159317136, "clip_ratio/region_mean": 0.07041908311657608, "reward_total_mean": 0.7958292365074158, "reward_meter_mean": 0.989596426486969, "reward_meter_std": 0.007186277769505978, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7901089191436768, "reward_repeat_soft_std": 0.058200206607580185, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.7958292365074158, "reward_total_composite_std": 0.04314915090799332} {"timestamp_utc": "2026-04-13T05:46:25Z", "mode": "train", "global_step": 2959, "epoch": 0.2972375690607735, "loss": 0.006, "grad_norm": 4.851568222045898, "learning_rate": 1.0363636363636365e-06, "num_tokens": 5537652.0, "completions/mean_length": 134.75, "completions/min_length": 131.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 134.75, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.9918241500854492, "rewards/meter/std": 0.0032640490680933, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8145939707756042, "rewards/repeat_soft/std": 0.060038018971681595, "rewards/judge_quality/mean": 0.6399999856948853, "rewards/judge_quality/std": 0.24331051111221313, "rewards/total_composite/mean": 0.8697803020477295, "rewards/total_composite/std": 0.07373517006635666, "reward": 0.8697803020477295, "reward_std": 0.07373517006635666, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06229768693447113, "sampling/sampling_logp_difference/max": 1.813847303390503, "sampling/importance_sampling_ratio/min": 0.16302572190761566, "sampling/importance_sampling_ratio/mean": 1.000590205192566, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2880987860262394, "clip_ratio/low_mean": 0.033367812633514404, "clip_ratio/low_min": 0.033367812633514404, "clip_ratio/high_mean": 0.024250525515526533, "clip_ratio/high_max": 0.024250525515526533, "clip_ratio/region_mean": 0.05761833814904094, "reward_total_mean": 0.8697803020477295, "reward_meter_mean": 0.9918241500854492, "reward_meter_std": 0.0032640490680933, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8145939707756042, "reward_repeat_soft_std": 0.060038018971681595, "reward_judge_quality_mean": 0.6399999856948853, "reward_judge_quality_std": 0.24331051111221313, "reward_total_composite_mean": 0.8697803020477295, "reward_total_composite_std": 0.07373517006635666} {"timestamp_utc": "2026-04-13T05:46:32Z", "mode": "train", "global_step": 2960, "epoch": 0.2973380210949272, "loss": -0.0019, "grad_norm": 7.309185981750488, "learning_rate": 1.0333333333333333e-06, "num_tokens": 5539820.0, "completions/mean_length": 100.0, "completions/min_length": 92.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.0, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.9756019115447998, "rewards/meter/std": 0.01948678493499756, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9064444303512573, "rewards/repeat_soft/std": 0.08656104654073715, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.20860078930854797, "rewards/total_composite/mean": 0.8069778680801392, "rewards/total_composite/std": 0.06588814407587051, "reward": 0.8069778680801392, "reward_std": 0.06588813662528992, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09263099730014801, "sampling/sampling_logp_difference/max": 1.1360344886779785, "sampling/importance_sampling_ratio/min": 0.3210897743701935, "sampling/importance_sampling_ratio/mean": 1.0097733736038208, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4792087823152542, "clip_ratio/low_mean": 0.049100569915026426, "clip_ratio/low_min": 0.049100569915026426, "clip_ratio/high_mean": 0.03331885952502489, "clip_ratio/high_max": 0.03331885952502489, "clip_ratio/region_mean": 0.08241942944005132, "reward_total_mean": 0.8069778680801392, "reward_meter_mean": 0.9756019115447998, "reward_meter_std": 0.01948678493499756, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9064444303512573, "reward_repeat_soft_std": 0.08656104654073715, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.20860078930854797, "reward_total_composite_mean": 0.8069778680801392, "reward_total_composite_std": 0.06588814407587051} {"timestamp_utc": "2026-04-13T05:46:39Z", "mode": "train", "global_step": 2961, "epoch": 0.29743847312908084, "loss": 0.0324, "grad_norm": 7.895133018493652, "learning_rate": 1.0303030303030304e-06, "num_tokens": 5542245.0, "completions/mean_length": 113.125, "completions/min_length": 105.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.125, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.7148743271827698, "rewards/meter/std": 0.25316041707992554, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9342986345291138, "rewards/repeat_soft/std": 0.050421342253685, "rewards/judge_quality/mean": 0.23375000059604645, "rewards/judge_quality/std": 0.09006941318511963, "rewards/total_composite/mean": 0.6352483034133911, "rewards/total_composite/std": 0.11865576356649399, "reward": 0.6352483034133911, "reward_std": 0.11865575611591339, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09733887761831284, "sampling/sampling_logp_difference/max": 2.240222930908203, "sampling/importance_sampling_ratio/min": 0.10643477737903595, "sampling/importance_sampling_ratio/mean": 0.9974368214607239, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4263001009821892, "clip_ratio/low_mean": 0.025418367702513933, "clip_ratio/low_min": 0.025418367702513933, "clip_ratio/high_mean": 0.04678329499438405, "clip_ratio/high_max": 0.04678329499438405, "clip_ratio/region_mean": 0.07220166269689798, "reward_total_mean": 0.6352483034133911, "reward_meter_mean": 0.7148743271827698, "reward_meter_std": 0.25316041707992554, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9342986345291138, "reward_repeat_soft_std": 0.050421342253685, "reward_judge_quality_mean": 0.23375000059604645, "reward_judge_quality_std": 0.09006941318511963, "reward_total_composite_mean": 0.6352483034133911, "reward_total_composite_std": 0.11865576356649399} {"timestamp_utc": "2026-04-13T05:46:46Z", "mode": "train", "global_step": 2962, "epoch": 0.29753892516323455, "loss": 0.0074, "grad_norm": 11.672962188720703, "learning_rate": 1.0272727272727274e-06, "num_tokens": 5543917.0, "completions/mean_length": 43.0, "completions/min_length": 40.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.0, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.9294984340667725, "rewards/meter/std": 0.04548434540629387, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9276872873306274, "rewards/repeat_soft/std": 0.07984476536512375, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7806680202484131, "rewards/total_composite/std": 0.020928001031279564, "reward": 0.7806680202484131, "reward_std": 0.020927993580698967, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07240515947341919, "sampling/sampling_logp_difference/max": 1.3342781066894531, "sampling/importance_sampling_ratio/min": 0.26334822177886963, "sampling/importance_sampling_ratio/mean": 1.0023572444915771, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3564946800470352, "clip_ratio/low_mean": 0.04671351891011, "clip_ratio/low_min": 0.04671351891011, "clip_ratio/high_mean": 0.029166667023673654, "clip_ratio/high_max": 0.029166667023673654, "clip_ratio/region_mean": 0.07588018593378365, "reward_total_mean": 0.7806680202484131, "reward_meter_mean": 0.9294984340667725, "reward_meter_std": 0.04548434540629387, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9276872873306274, "reward_repeat_soft_std": 0.07984476536512375, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7806680202484131, "reward_total_composite_std": 0.020928001031279564} {"timestamp_utc": "2026-04-13T05:46:52Z", "mode": "train", "global_step": 2963, "epoch": 0.29763937719738826, "loss": 0.0204, "grad_norm": 9.710457801818848, "learning_rate": 1.0242424242424242e-06, "num_tokens": 5545565.0, "completions/mean_length": 45.0, "completions/min_length": 40.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.0, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.9690317511558533, "rewards/meter/std": 0.05361192673444748, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.844587504863739, "rewards/repeat_soft/std": 0.065106600522995, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.18873640894889832, "rewards/total_composite/mean": 0.828773021697998, "rewards/total_composite/std": 0.05066449195146561, "reward": 0.828773021697998, "reward_std": 0.05066449195146561, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09703899174928665, "sampling/sampling_logp_difference/max": 2.5275778770446777, "sampling/importance_sampling_ratio/min": 0.07985220104455948, "sampling/importance_sampling_ratio/mean": 1.0070902109146118, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45457877963781357, "clip_ratio/low_mean": 0.07195135415531695, "clip_ratio/low_min": 0.07195135415531695, "clip_ratio/high_mean": 0.0058139534667134285, "clip_ratio/high_max": 0.0058139534667134285, "clip_ratio/region_mean": 0.07776530762203038, "reward_total_mean": 0.828773021697998, "reward_meter_mean": 0.9690317511558533, "reward_meter_std": 0.05361192673444748, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.844587504863739, "reward_repeat_soft_std": 0.065106600522995, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.18873640894889832, "reward_total_composite_mean": 0.828773021697998, "reward_total_composite_std": 0.05066449195146561} {"timestamp_utc": "2026-04-13T05:46:59Z", "mode": "train", "global_step": 2964, "epoch": 0.29773982923154196, "loss": -0.045, "grad_norm": 11.488801002502441, "learning_rate": 1.0212121212121213e-06, "num_tokens": 5547214.0, "completions/mean_length": 33.125, "completions/min_length": 28.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.125, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.8177497982978821, "rewards/meter/std": 0.33898067474365234, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9622748494148254, "rewards/repeat_soft/std": 0.0006367546156980097, "rewards/judge_quality/mean": 0.8575000166893005, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.8714649081230164, "rewards/total_composite/std": 0.20447571575641632, "reward": 0.8714649081230164, "reward_std": 0.20447570085525513, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09285153448581696, "sampling/sampling_logp_difference/max": 1.2033491134643555, "sampling/importance_sampling_ratio/min": 0.3001871705055237, "sampling/importance_sampling_ratio/mean": 1.0004459619522095, "sampling/importance_sampling_ratio/max": 1.7220011949539185, "entropy": 0.6089639738202095, "clip_ratio/low_mean": 0.008928571827709675, "clip_ratio/low_min": 0.008928571827709675, "clip_ratio/high_mean": 0.07974499557167292, "clip_ratio/high_max": 0.07974499557167292, "clip_ratio/region_mean": 0.08867356739938259, "reward_total_mean": 0.8714649081230164, "reward_meter_mean": 0.8177497982978821, "reward_meter_std": 0.33898067474365234, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9622748494148254, "reward_repeat_soft_std": 0.0006367546156980097, "reward_judge_quality_mean": 0.8575000166893005, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.8714649081230164, "reward_total_composite_std": 0.20447571575641632} {"timestamp_utc": "2026-04-13T05:47:11Z", "mode": "train", "global_step": 2965, "epoch": 0.2978402812656956, "loss": -0.1532, "grad_norm": 1.6521289348602295, "learning_rate": 1.0181818181818183e-06, "num_tokens": 5549057.0, "completions/mean_length": 180.375, "completions/min_length": 66.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 69.83333587646484, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.7748757600784302, "rewards/meter/std": 0.4007801413536072, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9761244058609009, "rewards/repeat_soft/std": 0.027295337989926338, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.17639240622520447, "rewards/total_composite/mean": 0.6167088747024536, "rewards/total_composite/std": 0.38064518570899963, "reward": 0.6167088747024536, "reward_std": 0.38064518570899963, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09999796748161316, "sampling/sampling_logp_difference/max": 1.3396964073181152, "sampling/importance_sampling_ratio/min": 0.2619251608848572, "sampling/importance_sampling_ratio/mean": 1.0290602445602417, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3892729952931404, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.04651618911884725, "clip_ratio/high_max": 0.04651618911884725, "clip_ratio/region_mean": 0.04651618911884725, "reward_total_mean": 0.6167088747024536, "reward_meter_mean": 0.7748757600784302, "reward_meter_std": 0.4007801413536072, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9761244058609009, "reward_repeat_soft_std": 0.027295337989926338, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.17639240622520447, "reward_total_composite_mean": 0.6167088747024536, "reward_total_composite_std": 0.38064518570899963} {"timestamp_utc": "2026-04-13T05:47:18Z", "mode": "train", "global_step": 2966, "epoch": 0.2979407332998493, "loss": 0.0291, "grad_norm": 9.937371253967285, "learning_rate": 1.0151515151515152e-06, "num_tokens": 5550698.0, "completions/mean_length": 62.125, "completions/min_length": 56.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.125, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9895007610321045, "rewards/meter/std": 0.0031189224682748318, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9905010461807251, "rewards/repeat_soft/std": 0.005832768511027098, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.263435423374176, "rewards/total_composite/mean": 0.896450400352478, "rewards/total_composite/std": 0.07847026735544205, "reward": 0.896450400352478, "reward_std": 0.07847028225660324, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08543248474597931, "sampling/sampling_logp_difference/max": 1.8757764101028442, "sampling/importance_sampling_ratio/min": 0.15323594212532043, "sampling/importance_sampling_ratio/mean": 1.005399227142334, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4025779590010643, "clip_ratio/low_mean": 0.03308332711458206, "clip_ratio/low_min": 0.03308332711458206, "clip_ratio/high_mean": 0.03668239340186119, "clip_ratio/high_max": 0.03668239340186119, "clip_ratio/region_mean": 0.06976572051644325, "reward_total_mean": 0.896450400352478, "reward_meter_mean": 0.9895007610321045, "reward_meter_std": 0.0031189224682748318, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9905010461807251, "reward_repeat_soft_std": 0.005832768511027098, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.263435423374176, "reward_total_composite_mean": 0.896450400352478, "reward_total_composite_std": 0.07847026735544205} {"timestamp_utc": "2026-04-13T05:47:24Z", "mode": "train", "global_step": 2967, "epoch": 0.29804118533400303, "loss": 0.0594, "grad_norm": 12.704472541809082, "learning_rate": 1.0121212121212122e-06, "num_tokens": 5552209.0, "completions/mean_length": 33.875, "completions/min_length": 30.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.875, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.9862372875213623, "rewards/meter/std": 0.01102246344089508, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9598589539527893, "rewards/repeat_soft/std": 0.004916208330541849, "rewards/judge_quality/mean": 0.5012500286102295, "rewards/judge_quality/std": 0.16974246501922607, "rewards/total_composite/mean": 0.8401677012443542, "rewards/total_composite/std": 0.05187404155731201, "reward": 0.8401677012443542, "reward_std": 0.05187402293086052, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12357749789953232, "sampling/sampling_logp_difference/max": 1.8039411306381226, "sampling/importance_sampling_ratio/min": 0.16464871168136597, "sampling/importance_sampling_ratio/mean": 0.9989071488380432, "sampling/importance_sampling_ratio/max": 1.839284896850586, "entropy": 0.6438338309526443, "clip_ratio/low_mean": 0.10371574200689793, "clip_ratio/low_min": 0.10371574200689793, "clip_ratio/high_mean": 0.012500000186264515, "clip_ratio/high_max": 0.012500000186264515, "clip_ratio/region_mean": 0.11621574219316244, "reward_total_mean": 0.8401677012443542, "reward_meter_mean": 0.9862372875213623, "reward_meter_std": 0.01102246344089508, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9598589539527893, "reward_repeat_soft_std": 0.004916208330541849, "reward_judge_quality_mean": 0.5012500286102295, "reward_judge_quality_std": 0.16974246501922607, "reward_total_composite_mean": 0.8401677012443542, "reward_total_composite_std": 0.05187404155731201} {"timestamp_utc": "2026-04-13T05:47:35Z", "mode": "train", "global_step": 2968, "epoch": 0.2981416373681567, "loss": -0.1253, "grad_norm": 1.3606377840042114, "learning_rate": 1.0090909090909092e-06, "num_tokens": 5553762.0, "completions/mean_length": 102.125, "completions/min_length": 41.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 43.57143020629883, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.860670804977417, "rewards/meter/std": 0.304593563079834, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9588666558265686, "rewards/repeat_soft/std": 0.032163456082344055, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.23439893126487732, "rewards/total_composite/mean": 0.7260405421257019, "rewards/total_composite/std": 0.2973344624042511, "reward": 0.7260405421257019, "reward_std": 0.2973344326019287, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08156950026750565, "sampling/sampling_logp_difference/max": 1.5375630855560303, "sampling/importance_sampling_ratio/min": 0.21490417420864105, "sampling/importance_sampling_ratio/mean": 0.9742173552513123, "sampling/importance_sampling_ratio/max": 1.4366979598999023, "entropy": 0.27850085124373436, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.07731056399643421, "clip_ratio/high_max": 0.07731056399643421, "clip_ratio/region_mean": 0.07731056399643421, "reward_total_mean": 0.7260405421257019, "reward_meter_mean": 0.860670804977417, "reward_meter_std": 0.304593563079834, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9588666558265686, "reward_repeat_soft_std": 0.032163456082344055, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.23439893126487732, "reward_total_composite_mean": 0.7260405421257019, "reward_total_composite_std": 0.2973344624042511} {"timestamp_utc": "2026-04-13T05:47:43Z", "mode": "train", "global_step": 2969, "epoch": 0.2982420894023104, "loss": 0.0085, "grad_norm": 8.067988395690918, "learning_rate": 1.006060606060606e-06, "num_tokens": 5555349.0, "completions/mean_length": 47.375, "completions/min_length": 40.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.375, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.9688791036605835, "rewards/meter/std": 0.03318021446466446, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9791945219039917, "rewards/repeat_soft/std": 0.011647253297269344, "rewards/judge_quality/mean": 0.8575000166893005, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.9411650896072388, "rewards/total_composite/std": 0.05672057345509529, "reward": 0.9411650896072388, "reward_std": 0.05672057345509529, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09939691424369812, "sampling/sampling_logp_difference/max": 1.348381519317627, "sampling/importance_sampling_ratio/min": 0.25966015458106995, "sampling/importance_sampling_ratio/mean": 1.0195558071136475, "sampling/importance_sampling_ratio/max": 1.9022397994995117, "entropy": 0.4933800511062145, "clip_ratio/low_mean": 0.025255102664232254, "clip_ratio/low_min": 0.025255102664232254, "clip_ratio/high_mean": 0.06514179660007358, "clip_ratio/high_max": 0.06514179660007358, "clip_ratio/region_mean": 0.09039689926430583, "reward_total_mean": 0.9411650896072388, "reward_meter_mean": 0.9688791036605835, "reward_meter_std": 0.03318021446466446, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9791945219039917, "reward_repeat_soft_std": 0.011647253297269344, "reward_judge_quality_mean": 0.8575000166893005, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.9411650896072388, "reward_total_composite_std": 0.05672057345509529} {"timestamp_utc": "2026-04-13T05:47:51Z", "mode": "train", "global_step": 2970, "epoch": 0.2983425414364641, "loss": 0.0277, "grad_norm": 4.905091762542725, "learning_rate": 1.0030303030303031e-06, "num_tokens": 5558350.0, "completions/mean_length": 174.125, "completions/min_length": 164.0, "completions/max_length": 188.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 174.125, "completions/min_terminated_length": 164.0, "completions/max_terminated_length": 188.0, "rewards/meter/mean": 0.9909360408782959, "rewards/meter/std": 0.0029742689803242683, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8123766183853149, "rewards/repeat_soft/std": 0.07790201902389526, "rewards/judge_quality/mean": 0.23000000417232513, "rewards/judge_quality/std": 0.12224097549915314, "rewards/total_composite/mean": 0.7461589574813843, "rewards/total_composite/std": 0.03726159408688545, "reward": 0.7461589574813843, "reward_std": 0.037261586636304855, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0745457336306572, "sampling/sampling_logp_difference/max": 2.6468048095703125, "sampling/importance_sampling_ratio/min": 0.0708773136138916, "sampling/importance_sampling_ratio/mean": 1.0009602308273315, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3535380996763706, "clip_ratio/low_mean": 0.04840483609586954, "clip_ratio/low_min": 0.04840483609586954, "clip_ratio/high_mean": 0.019275210332125425, "clip_ratio/high_max": 0.019275210332125425, "clip_ratio/region_mean": 0.06768004642799497, "reward_total_mean": 0.7461589574813843, "reward_meter_mean": 0.9909360408782959, "reward_meter_std": 0.0029742689803242683, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8123766183853149, "reward_repeat_soft_std": 0.07790201902389526, "reward_judge_quality_mean": 0.23000000417232513, "reward_judge_quality_std": 0.12224097549915314, "reward_total_composite_mean": 0.7461589574813843, "reward_total_composite_std": 0.03726159408688545} {"timestamp_utc": "2026-04-13T05:48:02Z", "mode": "train", "global_step": 2971, "epoch": 0.29844299347061776, "loss": -0.2165, "grad_norm": 1.2345492839813232, "learning_rate": 1.0000000000000002e-06, "num_tokens": 5560558.0, "completions/mean_length": 172.0, "completions/min_length": 116.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 123.42857360839844, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.8750044703483582, "rewards/meter/std": 0.24986883997917175, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9451867341995239, "rewards/repeat_soft/std": 0.04133282229304314, "rewards/judge_quality/mean": 0.2487499862909317, "rewards/judge_quality/std": 0.1364276260137558, "rewards/total_composite/mean": 0.6663497686386108, "rewards/total_composite/std": 0.27059048414230347, "reward": 0.6663497686386108, "reward_std": 0.27059048414230347, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07564030587673187, "sampling/sampling_logp_difference/max": 1.758089542388916, "sampling/importance_sampling_ratio/min": 0.17237386107444763, "sampling/importance_sampling_ratio/mean": 0.9953554272651672, "sampling/importance_sampling_ratio/max": 1.8869447708129883, "entropy": 0.3610876612365246, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06340434309095144, "clip_ratio/high_max": 0.06340434309095144, "clip_ratio/region_mean": 0.06340434309095144, "reward_total_mean": 0.6663497686386108, "reward_meter_mean": 0.8750044703483582, "reward_meter_std": 0.24986883997917175, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9451867341995239, "reward_repeat_soft_std": 0.04133282229304314, "reward_judge_quality_mean": 0.2487499862909317, "reward_judge_quality_std": 0.1364276260137558, "reward_total_composite_mean": 0.6663497686386108, "reward_total_composite_std": 0.27059048414230347} {"timestamp_utc": "2026-04-13T05:48:08Z", "mode": "train", "global_step": 2972, "epoch": 0.29854344550477147, "loss": -0.0012, "grad_norm": 5.584332466125488, "learning_rate": 9.96969696969697e-07, "num_tokens": 5562086.0, "completions/mean_length": 22.0, "completions/min_length": 22.0, "completions/max_length": 22.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.0, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 22.0, "rewards/meter/mean": 0.9569528698921204, "rewards/meter/std": 0.000598653859924525, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.25, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7518787980079651, "rewards/total_composite/std": 0.0002694026625249535, "reward": 0.7518787980079651, "reward_std": 0.0002694026625249535, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017133712768554688, "sampling/sampling_logp_difference/max": 0.3438436985015869, "sampling/importance_sampling_ratio/min": 0.7090397477149963, "sampling/importance_sampling_ratio/mean": 1.0009825229644775, "sampling/importance_sampling_ratio/max": 1.204524040222168, "entropy": 0.15260382182896137, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.017045455053448677, "clip_ratio/high_max": 0.017045455053448677, "clip_ratio/region_mean": 0.017045455053448677, "reward_total_mean": 0.7518787980079651, "reward_meter_mean": 0.9569528698921204, "reward_meter_std": 0.000598653859924525, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.25, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7518787980079651, "reward_total_composite_std": 0.0002694026625249535} {"timestamp_utc": "2026-04-13T05:48:14Z", "mode": "train", "global_step": 2973, "epoch": 0.2986438975389252, "loss": 0.0091, "grad_norm": 17.762943267822266, "learning_rate": 9.93939393939394e-07, "num_tokens": 5563496.0, "completions/mean_length": 28.25, "completions/min_length": 24.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.25, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9757226705551147, "rewards/meter/std": 0.01842680387198925, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9501981735229492, "rewards/repeat_soft/std": 0.03479473292827606, "rewards/judge_quality/mean": 0.39750000834465027, "rewards/judge_quality/std": 0.10110107809305191, "rewards/total_composite/mean": 0.8033449649810791, "rewards/total_composite/std": 0.029211973771452904, "reward": 0.8033449649810791, "reward_std": 0.02921196259558201, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08568035066127777, "sampling/sampling_logp_difference/max": 1.1337909698486328, "sampling/importance_sampling_ratio/min": 0.3218109607696533, "sampling/importance_sampling_ratio/mean": 0.9961305856704712, "sampling/importance_sampling_ratio/max": 1.5311850309371948, "entropy": 0.5338125750422478, "clip_ratio/low_mean": 0.03575989790260792, "clip_ratio/low_min": 0.03575989790260792, "clip_ratio/high_mean": 0.06274240487255156, "clip_ratio/high_max": 0.06274240487255156, "clip_ratio/region_mean": 0.09850230277515948, "reward_total_mean": 0.8033449649810791, "reward_meter_mean": 0.9757226705551147, "reward_meter_std": 0.01842680387198925, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9501981735229492, "reward_repeat_soft_std": 0.03479473292827606, "reward_judge_quality_mean": 0.39750000834465027, "reward_judge_quality_std": 0.10110107809305191, "reward_total_composite_mean": 0.8033449649810791, "reward_total_composite_std": 0.029211973771452904} {"timestamp_utc": "2026-04-13T05:48:25Z", "mode": "train", "global_step": 2974, "epoch": 0.2987443495730789, "loss": -0.154, "grad_norm": 1.9390344619750977, "learning_rate": 9.90909090909091e-07, "num_tokens": 5565328.0, "completions/mean_length": 116.0, "completions/min_length": 54.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 59.42857360839844, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.8655053973197937, "rewards/meter/std": 0.1947968304157257, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9812927842140198, "rewards/repeat_soft/std": 0.012936427257955074, "rewards/judge_quality/mean": 0.4050000011920929, "rewards/judge_quality/std": 0.2523036599159241, "rewards/total_composite/mean": 0.6956750750541687, "rewards/total_composite/std": 0.29536378383636475, "reward": 0.6956750750541687, "reward_std": 0.29536375403404236, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1034252941608429, "sampling/sampling_logp_difference/max": 1.7186291217803955, "sampling/importance_sampling_ratio/min": 0.17931179702281952, "sampling/importance_sampling_ratio/mean": 1.010888695716858, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.38944845274090767, "clip_ratio/low_mean": 0.01315789483487606, "clip_ratio/low_min": 0.01315789483487606, "clip_ratio/high_mean": 0.07593737402930856, "clip_ratio/high_max": 0.07593737402930856, "clip_ratio/region_mean": 0.08909526886418462, "reward_total_mean": 0.6956750750541687, "reward_meter_mean": 0.8655053973197937, "reward_meter_std": 0.1947968304157257, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9812927842140198, "reward_repeat_soft_std": 0.012936427257955074, "reward_judge_quality_mean": 0.4050000011920929, "reward_judge_quality_std": 0.2523036599159241, "reward_total_composite_mean": 0.6956750750541687, "reward_total_composite_std": 0.29536378383636475} {"timestamp_utc": "2026-04-13T05:48:31Z", "mode": "train", "global_step": 2975, "epoch": 0.29884480160723254, "loss": 0.0161, "grad_norm": 12.002836227416992, "learning_rate": 9.87878787878788e-07, "num_tokens": 5566811.0, "completions/mean_length": 31.375, "completions/min_length": 28.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.375, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9899876713752747, "rewards/meter/std": 0.00508477445691824, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9602810144424438, "rewards/repeat_soft/std": 0.006276086904108524, "rewards/judge_quality/mean": 0.44624999165534973, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8253975510597229, "rewards/total_composite/std": 0.003816203447058797, "reward": 0.8253975510597229, "reward_std": 0.003816187847405672, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07486128062009811, "sampling/sampling_logp_difference/max": 1.241410732269287, "sampling/importance_sampling_ratio/min": 0.2889762818813324, "sampling/importance_sampling_ratio/mean": 1.0011824369430542, "sampling/importance_sampling_ratio/max": 1.8805824518203735, "entropy": 0.4146346226334572, "clip_ratio/low_mean": 0.03132511069998145, "clip_ratio/low_min": 0.03132511069998145, "clip_ratio/high_mean": 0.057160365860909224, "clip_ratio/high_max": 0.057160365860909224, "clip_ratio/region_mean": 0.08848547656089067, "reward_total_mean": 0.8253975510597229, "reward_meter_mean": 0.9899876713752747, "reward_meter_std": 0.00508477445691824, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9602810144424438, "reward_repeat_soft_std": 0.006276086904108524, "reward_judge_quality_mean": 0.44624999165534973, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8253975510597229, "reward_total_composite_std": 0.003816203447058797} {"timestamp_utc": "2026-04-13T05:48:38Z", "mode": "train", "global_step": 2976, "epoch": 0.29894525364138624, "loss": 0.047, "grad_norm": 9.028831481933594, "learning_rate": 9.84848484848485e-07, "num_tokens": 5568513.0, "completions/mean_length": 63.75, "completions/min_length": 58.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.75, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.7205665111541748, "rewards/meter/std": 0.28043070435523987, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9899735450744629, "rewards/repeat_soft/std": 0.010403034277260303, "rewards/judge_quality/mean": 0.6599999666213989, "rewards/judge_quality/std": 0.2570158541202545, "rewards/total_composite/mean": 0.7712522745132446, "rewards/total_composite/std": 0.16571196913719177, "reward": 0.7712522745132446, "reward_std": 0.16571195423603058, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09550134092569351, "sampling/sampling_logp_difference/max": 1.708512783050537, "sampling/importance_sampling_ratio/min": 0.18113498389720917, "sampling/importance_sampling_ratio/mean": 0.9994534850120544, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5011239908635616, "clip_ratio/low_mean": 0.035175614058971405, "clip_ratio/low_min": 0.035175614058971405, "clip_ratio/high_mean": 0.07625904306769371, "clip_ratio/high_max": 0.07625904306769371, "clip_ratio/region_mean": 0.11143465712666512, "reward_total_mean": 0.7712522745132446, "reward_meter_mean": 0.7205665111541748, "reward_meter_std": 0.28043070435523987, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9899735450744629, "reward_repeat_soft_std": 0.010403034277260303, "reward_judge_quality_mean": 0.6599999666213989, "reward_judge_quality_std": 0.2570158541202545, "reward_total_composite_mean": 0.7712522745132446, "reward_total_composite_std": 0.16571196913719177} {"timestamp_utc": "2026-04-13T05:48:45Z", "mode": "train", "global_step": 2977, "epoch": 0.29904570567553995, "loss": 0.0287, "grad_norm": 10.380030632019043, "learning_rate": 9.818181818181818e-07, "num_tokens": 5570157.0, "completions/mean_length": 53.5, "completions/min_length": 48.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.5, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9852970838546753, "rewards/meter/std": 0.006312056910246611, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9889814853668213, "rewards/repeat_soft/std": 0.009862289763987064, "rewards/judge_quality/mean": 0.7362500429153442, "rewards/judge_quality/std": 0.25376805663108826, "rewards/total_composite/mean": 0.9131568670272827, "rewards/total_composite/std": 0.07736536115407944, "reward": 0.9131568670272827, "reward_std": 0.07736535370349884, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07505526393651962, "sampling/sampling_logp_difference/max": 1.1849522590637207, "sampling/importance_sampling_ratio/min": 0.30576080083847046, "sampling/importance_sampling_ratio/mean": 1.0003772974014282, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.35377485677599907, "clip_ratio/low_mean": 0.020803289022296667, "clip_ratio/low_min": 0.020803289022296667, "clip_ratio/high_mean": 0.031005950178951025, "clip_ratio/high_max": 0.031005950178951025, "clip_ratio/region_mean": 0.05180923920124769, "reward_total_mean": 0.9131568670272827, "reward_meter_mean": 0.9852970838546753, "reward_meter_std": 0.006312056910246611, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9889814853668213, "reward_repeat_soft_std": 0.009862289763987064, "reward_judge_quality_mean": 0.7362500429153442, "reward_judge_quality_std": 0.25376805663108826, "reward_total_composite_mean": 0.9131568670272827, "reward_total_composite_std": 0.07736536115407944} {"timestamp_utc": "2026-04-13T05:48:52Z", "mode": "train", "global_step": 2978, "epoch": 0.2991461577096936, "loss": -0.0025, "grad_norm": 5.545828342437744, "learning_rate": 9.787878787878788e-07, "num_tokens": 5572244.0, "completions/mean_length": 105.875, "completions/min_length": 97.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.875, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.9860951900482178, "rewards/meter/std": 0.004379128571599722, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9779505729675293, "rewards/repeat_soft/std": 0.01805206574499607, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.8362878561019897, "rewards/total_composite/std": 0.053976088762283325, "reward": 0.8362878561019897, "reward_std": 0.05397608131170273, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06535305082798004, "sampling/sampling_logp_difference/max": 3.0498671531677246, "sampling/importance_sampling_ratio/min": 0.0473652146756649, "sampling/importance_sampling_ratio/mean": 1.0022234916687012, "sampling/importance_sampling_ratio/max": 1.7940611839294434, "entropy": 0.31342941150069237, "clip_ratio/low_mean": 0.052398128900676966, "clip_ratio/low_min": 0.052398128900676966, "clip_ratio/high_mean": 0.006756756920367479, "clip_ratio/high_max": 0.006756756920367479, "clip_ratio/region_mean": 0.059154885821044445, "reward_total_mean": 0.8362878561019897, "reward_meter_mean": 0.9860951900482178, "reward_meter_std": 0.004379128571599722, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9779505729675293, "reward_repeat_soft_std": 0.01805206574499607, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.8362878561019897, "reward_total_composite_std": 0.053976088762283325} {"timestamp_utc": "2026-04-13T05:48:58Z", "mode": "train", "global_step": 2979, "epoch": 0.2992466097438473, "loss": 0.0106, "grad_norm": 11.730058670043945, "learning_rate": 9.757575757575759e-07, "num_tokens": 5573905.0, "completions/mean_length": 50.625, "completions/min_length": 44.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.625, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.6245219707489014, "rewards/meter/std": 0.25448864698410034, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9643666744232178, "rewards/repeat_soft/std": 0.03061577118933201, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.12906256318092346, "rewards/total_composite/mean": 0.6594715714454651, "rewards/total_composite/std": 0.13807697594165802, "reward": 0.6594715714454651, "reward_std": 0.13807696104049683, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10324420034885406, "sampling/sampling_logp_difference/max": 1.5343955755233765, "sampling/importance_sampling_ratio/min": 0.21558596193790436, "sampling/importance_sampling_ratio/mean": 0.9990389943122864, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45142338797450066, "clip_ratio/low_mean": 0.03774951305240393, "clip_ratio/low_min": 0.03774951305240393, "clip_ratio/high_mean": 0.03889483027160168, "clip_ratio/high_max": 0.03889483027160168, "clip_ratio/region_mean": 0.0766443433240056, "reward_total_mean": 0.6594715714454651, "reward_meter_mean": 0.6245219707489014, "reward_meter_std": 0.25448864698410034, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9643666744232178, "reward_repeat_soft_std": 0.03061577118933201, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.12906256318092346, "reward_total_composite_mean": 0.6594715714454651, "reward_total_composite_std": 0.13807697594165802} {"timestamp_utc": "2026-04-13T05:49:05Z", "mode": "train", "global_step": 2980, "epoch": 0.299347061778001, "loss": -0.0092, "grad_norm": 9.405289649963379, "learning_rate": 9.72727272727273e-07, "num_tokens": 5575681.0, "completions/mean_length": 54.0, "completions/min_length": 45.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9902579188346863, "rewards/meter/std": 0.00326842674985528, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9648735523223877, "rewards/repeat_soft/std": 0.027902567759156227, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8248534202575684, "rewards/total_composite/std": 0.005759947933256626, "reward": 0.8248534202575684, "reward_std": 0.005759951192885637, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10670540481805801, "sampling/sampling_logp_difference/max": 1.4616612195968628, "sampling/importance_sampling_ratio/min": 0.23185080289840698, "sampling/importance_sampling_ratio/mean": 1.0025509595870972, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44730718061327934, "clip_ratio/low_mean": 0.05610119178891182, "clip_ratio/low_min": 0.05610119178891182, "clip_ratio/high_mean": 0.05884688161313534, "clip_ratio/high_max": 0.05884688161313534, "clip_ratio/region_mean": 0.11494807340204716, "reward_total_mean": 0.8248534202575684, "reward_meter_mean": 0.9902579188346863, "reward_meter_std": 0.00326842674985528, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9648735523223877, "reward_repeat_soft_std": 0.027902567759156227, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8248534202575684, "reward_total_composite_std": 0.005759947933256626} {"timestamp_utc": "2026-04-13T05:49:12Z", "mode": "train", "global_step": 2981, "epoch": 0.2994475138121547, "loss": -0.0195, "grad_norm": 8.862262725830078, "learning_rate": 9.696969696969698e-07, "num_tokens": 5577486.0, "completions/mean_length": 56.625, "completions/min_length": 51.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.625, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9753900766372681, "rewards/meter/std": 0.03038530796766281, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9792236685752869, "rewards/repeat_soft/std": 0.025809839367866516, "rewards/judge_quality/mean": 0.47999998927116394, "rewards/judge_quality/std": 0.09754122048616409, "rewards/total_composite/mean": 0.8308478593826294, "rewards/total_composite/std": 0.03529951721429825, "reward": 0.8308478593826294, "reward_std": 0.035299502313137054, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10820525884628296, "sampling/sampling_logp_difference/max": 1.1371650695800781, "sampling/importance_sampling_ratio/min": 0.3207269608974457, "sampling/importance_sampling_ratio/mean": 1.0016381740570068, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5036165229976177, "clip_ratio/low_mean": 0.0690578855574131, "clip_ratio/low_min": 0.0690578855574131, "clip_ratio/high_mean": 0.025154695380479097, "clip_ratio/high_max": 0.025154695380479097, "clip_ratio/region_mean": 0.0942125809378922, "reward_total_mean": 0.8308478593826294, "reward_meter_mean": 0.9753900766372681, "reward_meter_std": 0.03038530796766281, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9792236685752869, "reward_repeat_soft_std": 0.025809839367866516, "reward_judge_quality_mean": 0.47999998927116394, "reward_judge_quality_std": 0.09754122048616409, "reward_total_composite_mean": 0.8308478593826294, "reward_total_composite_std": 0.03529951721429825} {"timestamp_utc": "2026-04-13T05:49:19Z", "mode": "train", "global_step": 2982, "epoch": 0.2995479658463084, "loss": 0.0355, "grad_norm": 5.1029181480407715, "learning_rate": 9.666666666666668e-07, "num_tokens": 5580032.0, "completions/mean_length": 131.25, "completions/min_length": 125.0, "completions/max_length": 147.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.25, "completions/min_terminated_length": 125.0, "completions/max_terminated_length": 147.0, "rewards/meter/mean": 0.9794935584068298, "rewards/meter/std": 0.026263222098350525, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.896114706993103, "rewards/repeat_soft/std": 0.031079640612006187, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.777133584022522, "rewards/total_composite/std": 0.03046722151339054, "reward": 0.777133584022522, "reward_std": 0.030467217788100243, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08723554015159607, "sampling/sampling_logp_difference/max": 1.361177921295166, "sampling/importance_sampling_ratio/min": 0.25635862350463867, "sampling/importance_sampling_ratio/mean": 1.0124614238739014, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40949783474206924, "clip_ratio/low_mean": 0.0485380778554827, "clip_ratio/low_min": 0.0485380778554827, "clip_ratio/high_mean": 0.024554687552154064, "clip_ratio/high_max": 0.024554687552154064, "clip_ratio/region_mean": 0.07309276540763676, "reward_total_mean": 0.777133584022522, "reward_meter_mean": 0.9794935584068298, "reward_meter_std": 0.026263222098350525, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.896114706993103, "reward_repeat_soft_std": 0.031079640612006187, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.777133584022522, "reward_total_composite_std": 0.03046722151339054} {"timestamp_utc": "2026-04-13T05:49:25Z", "mode": "train", "global_step": 2983, "epoch": 0.2996484178804621, "loss": 0.0104, "grad_norm": 10.487898826599121, "learning_rate": 9.636363636363636e-07, "num_tokens": 5581718.0, "completions/mean_length": 55.75, "completions/min_length": 51.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.75, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9646861553192139, "rewards/meter/std": 0.05977776274085045, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9907431602478027, "rewards/repeat_soft/std": 0.008609076961874962, "rewards/judge_quality/mean": 0.6187499761581421, "rewards/judge_quality/std": 0.24976776540279388, "rewards/total_composite/mean": 0.8688080906867981, "rewards/total_composite/std": 0.06664696335792542, "reward": 0.8688080906867981, "reward_std": 0.06664696335792542, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08234027773141861, "sampling/sampling_logp_difference/max": 1.0954147577285767, "sampling/importance_sampling_ratio/min": 0.33440089225769043, "sampling/importance_sampling_ratio/mean": 1.0238699913024902, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.441526859998703, "clip_ratio/low_mean": 0.054795289412140846, "clip_ratio/low_min": 0.054795289412140846, "clip_ratio/high_mean": 0.022219701670110226, "clip_ratio/high_max": 0.022219701670110226, "clip_ratio/region_mean": 0.07701499108225107, "reward_total_mean": 0.8688080906867981, "reward_meter_mean": 0.9646861553192139, "reward_meter_std": 0.05977776274085045, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9907431602478027, "reward_repeat_soft_std": 0.008609076961874962, "reward_judge_quality_mean": 0.6187499761581421, "reward_judge_quality_std": 0.24976776540279388, "reward_total_composite_mean": 0.8688080906867981, "reward_total_composite_std": 0.06664696335792542} {"timestamp_utc": "2026-04-13T05:49:32Z", "mode": "train", "global_step": 2984, "epoch": 0.29974886991461575, "loss": -0.0171, "grad_norm": 10.845376014709473, "learning_rate": 9.606060606060607e-07, "num_tokens": 5583297.0, "completions/mean_length": 52.375, "completions/min_length": 44.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.375, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.8542054891586304, "rewards/meter/std": 0.18442402780056, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9473631381988525, "rewards/repeat_soft/std": 0.03076014667749405, "rewards/judge_quality/mean": 0.8575000166893005, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.8863787651062012, "rewards/total_composite/std": 0.08437203615903854, "reward": 0.8863787651062012, "reward_std": 0.08437203615903854, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10809443145990372, "sampling/sampling_logp_difference/max": 1.7420437335968018, "sampling/importance_sampling_ratio/min": 0.17516204714775085, "sampling/importance_sampling_ratio/mean": 0.9890025854110718, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45414865016937256, "clip_ratio/low_mean": 0.04605146683752537, "clip_ratio/low_min": 0.04605146683752537, "clip_ratio/high_mean": 0.05169956432655454, "clip_ratio/high_max": 0.05169956432655454, "clip_ratio/region_mean": 0.0977510311640799, "reward_total_mean": 0.8863787651062012, "reward_meter_mean": 0.8542054891586304, "reward_meter_std": 0.18442402780056, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9473631381988525, "reward_repeat_soft_std": 0.03076014667749405, "reward_judge_quality_mean": 0.8575000166893005, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.8863787651062012, "reward_total_composite_std": 0.08437203615903854} {"timestamp_utc": "2026-04-13T05:49:39Z", "mode": "train", "global_step": 2985, "epoch": 0.29984932194876945, "loss": -0.024, "grad_norm": 4.063868999481201, "learning_rate": 9.575757575757577e-07, "num_tokens": 5585893.0, "completions/mean_length": 134.5, "completions/min_length": 107.0, "completions/max_length": 147.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 134.5, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 147.0, "rewards/meter/mean": 0.9883615970611572, "rewards/meter/std": 0.002527327975258231, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6620138883590698, "rewards/repeat_soft/std": 0.05591646581888199, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.7935266494750977, "rewards/total_composite/std": 0.03543936461210251, "reward": 0.7935266494750977, "reward_std": 0.03543936461210251, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.053613465279340744, "sampling/sampling_logp_difference/max": 3.1442694664001465, "sampling/importance_sampling_ratio/min": 0.04309839755296707, "sampling/importance_sampling_ratio/mean": 1.0039962530136108, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.24674266949295998, "clip_ratio/low_mean": 0.04216355876997113, "clip_ratio/low_min": 0.04216355876997113, "clip_ratio/high_mean": 0.007352941203862429, "clip_ratio/high_max": 0.007352941203862429, "clip_ratio/region_mean": 0.04951649997383356, "reward_total_mean": 0.7935266494750977, "reward_meter_mean": 0.9883615970611572, "reward_meter_std": 0.002527327975258231, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6620138883590698, "reward_repeat_soft_std": 0.05591646581888199, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.7935266494750977, "reward_total_composite_std": 0.03543936461210251} {"timestamp_utc": "2026-04-13T05:49:46Z", "mode": "train", "global_step": 2986, "epoch": 0.29994977398292316, "loss": 0.0315, "grad_norm": 8.785754203796387, "learning_rate": 9.545454545454548e-07, "num_tokens": 5587554.0, "completions/mean_length": 56.625, "completions/min_length": 51.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.625, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9841020107269287, "rewards/meter/std": 0.008267405442893505, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9765947461128235, "rewards/repeat_soft/std": 0.04103631153702736, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.8652553558349609, "rewards/total_composite/std": 0.0699370950460434, "reward": 0.8652553558349609, "reward_std": 0.0699370875954628, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08418867737054825, "sampling/sampling_logp_difference/max": 1.465388298034668, "sampling/importance_sampling_ratio/min": 0.2309882789850235, "sampling/importance_sampling_ratio/mean": 0.9953653812408447, "sampling/importance_sampling_ratio/max": 1.9144307374954224, "entropy": 0.4076434150338173, "clip_ratio/low_mean": 0.054496278055012226, "clip_ratio/low_min": 0.054496278055012226, "clip_ratio/high_mean": 0.024458204861730337, "clip_ratio/high_max": 0.024458204861730337, "clip_ratio/region_mean": 0.07895448291674256, "reward_total_mean": 0.8652553558349609, "reward_meter_mean": 0.9841020107269287, "reward_meter_std": 0.008267405442893505, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9765947461128235, "reward_repeat_soft_std": 0.04103631153702736, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.8652553558349609, "reward_total_composite_std": 0.0699370950460434} {"timestamp_utc": "2026-04-13T05:49:58Z", "mode": "train", "global_step": 2987, "epoch": 0.30005022601707687, "loss": -0.1714, "grad_norm": 1.6859650611877441, "learning_rate": 9.515151515151516e-07, "num_tokens": 5589383.0, "completions/mean_length": 128.625, "completions/min_length": 63.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 73.85714721679688, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9831241369247437, "rewards/meter/std": 0.027321094647049904, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.992339551448822, "rewards/repeat_soft/std": 0.007480784319341183, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.714828372001648, "rewards/total_composite/std": 0.28908124566078186, "reward": 0.714828372001648, "reward_std": 0.28908124566078186, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10756555944681168, "sampling/sampling_logp_difference/max": 2.052973747253418, "sampling/importance_sampling_ratio/min": 0.12835264205932617, "sampling/importance_sampling_ratio/mean": 1.004744291305542, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5285986922681332, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10141644533723593, "clip_ratio/high_max": 0.10141644533723593, "clip_ratio/region_mean": 0.10141644533723593, "reward_total_mean": 0.714828372001648, "reward_meter_mean": 0.9831241369247437, "reward_meter_std": 0.027321094647049904, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.992339551448822, "reward_repeat_soft_std": 0.007480784319341183, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.714828372001648, "reward_total_composite_std": 0.28908124566078186} {"timestamp_utc": "2026-04-13T05:50:05Z", "mode": "train", "global_step": 2988, "epoch": 0.3001506780512305, "loss": 0.0037, "grad_norm": 9.711652755737305, "learning_rate": 9.484848484848485e-07, "num_tokens": 5590937.0, "completions/mean_length": 44.25, "completions/min_length": 43.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.25, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.9646825790405273, "rewards/meter/std": 0.013903562910854816, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8864165544509888, "rewards/repeat_soft/std": 0.08947198837995529, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.798748791217804, "rewards/total_composite/std": 0.009591348469257355, "reward": 0.798748791217804, "reward_std": 0.009591348469257355, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04387456178665161, "sampling/sampling_logp_difference/max": 1.703268051147461, "sampling/importance_sampling_ratio/min": 0.1820874810218811, "sampling/importance_sampling_ratio/mean": 0.9953727126121521, "sampling/importance_sampling_ratio/max": 1.412367820739746, "entropy": 0.21737085096538067, "clip_ratio/low_mean": 0.02266414207406342, "clip_ratio/low_min": 0.02266414207406342, "clip_ratio/high_mean": 0.03397932858206332, "clip_ratio/high_max": 0.03397932858206332, "clip_ratio/region_mean": 0.05664347065612674, "reward_total_mean": 0.798748791217804, "reward_meter_mean": 0.9646825790405273, "reward_meter_std": 0.013903562910854816, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8864165544509888, "reward_repeat_soft_std": 0.08947198837995529, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.798748791217804, "reward_total_composite_std": 0.009591348469257355} {"timestamp_utc": "2026-04-13T05:50:16Z", "mode": "train", "global_step": 2989, "epoch": 0.30025113008538423, "loss": -0.1039, "grad_norm": 1.8749206066131592, "learning_rate": 9.454545454545455e-07, "num_tokens": 5592652.0, "completions/mean_length": 100.375, "completions/min_length": 35.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 41.57143020629883, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.7325649261474609, "rewards/meter/std": 0.3763197064399719, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9153746366500854, "rewards/repeat_soft/std": 0.07995446771383286, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.6006821393966675, "rewards/total_composite/std": 0.2946702837944031, "reward": 0.6006821393966675, "reward_std": 0.2946702539920807, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07462743669748306, "sampling/sampling_logp_difference/max": 1.539224624633789, "sampling/importance_sampling_ratio/min": 0.22013086080551147, "sampling/importance_sampling_ratio/mean": 0.9994422197341919, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.26893780939280987, "clip_ratio/low_mean": 0.024027293547987938, "clip_ratio/low_min": 0.024027293547987938, "clip_ratio/high_mean": 0.043295496609061956, "clip_ratio/high_max": 0.043295496609061956, "clip_ratio/region_mean": 0.0673227901570499, "reward_total_mean": 0.6006821393966675, "reward_meter_mean": 0.7325649261474609, "reward_meter_std": 0.3763197064399719, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9153746366500854, "reward_repeat_soft_std": 0.07995446771383286, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.6006821393966675, "reward_total_composite_std": 0.2946702837944031} {"timestamp_utc": "2026-04-13T05:50:24Z", "mode": "train", "global_step": 2990, "epoch": 0.30035158211953794, "loss": -0.0381, "grad_norm": 4.418584823608398, "learning_rate": 9.424242424242425e-07, "num_tokens": 5595670.0, "completions/mean_length": 194.25, "completions/min_length": 177.0, "completions/max_length": 213.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 194.25, "completions/min_terminated_length": 177.0, "completions/max_terminated_length": 213.0, "rewards/meter/mean": 0.9930227994918823, "rewards/meter/std": 0.0029159735422581434, "rewards/count_adherence/mean": 0.9166666269302368, "rewards/count_adherence/std": 0.0890870913863182, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8571102023124695, "rewards/repeat_soft/std": 0.04288844019174576, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7960712909698486, "rewards/total_composite/std": 0.014628508128225803, "reward": 0.7960712909698486, "reward_std": 0.014628510922193527, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06559725850820541, "sampling/sampling_logp_difference/max": 3.207681179046631, "sampling/importance_sampling_ratio/min": 0.04045030102133751, "sampling/importance_sampling_ratio/mean": 1.002389907836914, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3318157121539116, "clip_ratio/low_mean": 0.027267853263765574, "clip_ratio/low_min": 0.027267853263765574, "clip_ratio/high_mean": 0.02655445598065853, "clip_ratio/high_max": 0.02655445598065853, "clip_ratio/region_mean": 0.053822309244424105, "reward_total_mean": 0.7960712909698486, "reward_meter_mean": 0.9930227994918823, "reward_meter_std": 0.0029159735422581434, "reward_count_adherence_mean": 0.9166666269302368, "reward_count_adherence_std": 0.0890870913863182, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8571102023124695, "reward_repeat_soft_std": 0.04288844019174576, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7960712909698486, "reward_total_composite_std": 0.014628508128225803} {"timestamp_utc": "2026-04-13T05:50:30Z", "mode": "train", "global_step": 2991, "epoch": 0.3004520341536916, "loss": -0.0083, "grad_norm": 14.01867961883545, "learning_rate": 9.393939393939395e-07, "num_tokens": 5597073.0, "completions/mean_length": 26.375, "completions/min_length": 22.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.375, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.9873490333557129, "rewards/meter/std": 0.006079513113945723, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9497624635696411, "rewards/repeat_soft/std": 0.014834946021437645, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8220332860946655, "rewards/total_composite/std": 0.0054644751362502575, "reward": 0.8220332860946655, "reward_std": 0.005464466288685799, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08567684143781662, "sampling/sampling_logp_difference/max": 1.0796489715576172, "sampling/importance_sampling_ratio/min": 0.33971473574638367, "sampling/importance_sampling_ratio/mean": 1.0026715993881226, "sampling/importance_sampling_ratio/max": 1.637198805809021, "entropy": 0.4882727861404419, "clip_ratio/low_mean": 0.013141026254743338, "clip_ratio/low_min": 0.013141026254743338, "clip_ratio/high_mean": 0.05949012003839016, "clip_ratio/high_max": 0.05949012003839016, "clip_ratio/region_mean": 0.0726311462931335, "reward_total_mean": 0.8220332860946655, "reward_meter_mean": 0.9873490333557129, "reward_meter_std": 0.006079513113945723, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9497624635696411, "reward_repeat_soft_std": 0.014834946021437645, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8220332860946655, "reward_total_composite_std": 0.0054644751362502575} {"timestamp_utc": "2026-04-13T05:50:37Z", "mode": "train", "global_step": 2992, "epoch": 0.3005524861878453, "loss": 0.0405, "grad_norm": 6.706938743591309, "learning_rate": 9.363636363636365e-07, "num_tokens": 5599190.0, "completions/mean_length": 98.625, "completions/min_length": 89.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.625, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.9210655093193054, "rewards/meter/std": 0.0741414949297905, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9670059680938721, "rewards/repeat_soft/std": 0.027046864852309227, "rewards/judge_quality/mean": 0.5074999928474426, "rewards/judge_quality/std": 0.18077215552330017, "rewards/total_composite/mean": 0.8134300708770752, "rewards/total_composite/std": 0.043127965182065964, "reward": 0.8134300708770752, "reward_std": 0.043127965182065964, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11217209696769714, "sampling/sampling_logp_difference/max": 4.1335320472717285, "sampling/importance_sampling_ratio/min": 0.016026172786951065, "sampling/importance_sampling_ratio/mean": 1.007451057434082, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39667052775621414, "clip_ratio/low_mean": 0.038596793077886105, "clip_ratio/low_min": 0.038596793077886105, "clip_ratio/high_mean": 0.053689501248300076, "clip_ratio/high_max": 0.053689501248300076, "clip_ratio/region_mean": 0.09228629432618618, "reward_total_mean": 0.8134300708770752, "reward_meter_mean": 0.9210655093193054, "reward_meter_std": 0.0741414949297905, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9670059680938721, "reward_repeat_soft_std": 0.027046864852309227, "reward_judge_quality_mean": 0.5074999928474426, "reward_judge_quality_std": 0.18077215552330017, "reward_total_composite_mean": 0.8134300708770752, "reward_total_composite_std": 0.043127965182065964} {"timestamp_utc": "2026-04-13T05:50:48Z", "mode": "train", "global_step": 2993, "epoch": 0.300652938221999, "loss": -0.0975, "grad_norm": 1.3261058330535889, "learning_rate": 9.333333333333334e-07, "num_tokens": 5600833.0, "completions/mean_length": 89.375, "completions/min_length": 23.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 29.000001907348633, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.8699705600738525, "rewards/meter/std": 0.35134756565093994, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.38874998688697815, "rewards/judge_quality/std": 0.13767844438552856, "rewards/total_composite/mean": 0.7216798663139343, "rewards/total_composite/std": 0.2916436493396759, "reward": 0.7216798663139343, "reward_std": 0.2916436493396759, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11393612623214722, "sampling/sampling_logp_difference/max": 1.4696178436279297, "sampling/importance_sampling_ratio/min": 0.23001337051391602, "sampling/importance_sampling_ratio/mean": 1.0154926776885986, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5583476163446903, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09955350775271654, "clip_ratio/high_max": 0.09955350775271654, "clip_ratio/region_mean": 0.09955350775271654, "reward_total_mean": 0.7216798663139343, "reward_meter_mean": 0.8699705600738525, "reward_meter_std": 0.35134756565093994, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.38874998688697815, "reward_judge_quality_std": 0.13767844438552856, "reward_total_composite_mean": 0.7216798663139343, "reward_total_composite_std": 0.2916436493396759} {"timestamp_utc": "2026-04-13T05:50:59Z", "mode": "train", "global_step": 2994, "epoch": 0.30075339025615266, "loss": -0.1552, "grad_norm": 1.900068759918213, "learning_rate": 9.303030303030304e-07, "num_tokens": 5602499.0, "completions/mean_length": 121.25, "completions/min_length": 59.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 65.42857360839844, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.8232419490814209, "rewards/meter/std": 0.333120197057724, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9692974090576172, "rewards/repeat_soft/std": 0.026825375854969025, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.24479949474334717, "rewards/total_composite/mean": 0.7246972322463989, "rewards/total_composite/std": 0.298931747674942, "reward": 0.7246972322463989, "reward_std": 0.298931747674942, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11223641037940979, "sampling/sampling_logp_difference/max": 1.4623360633850098, "sampling/importance_sampling_ratio/min": 0.23169440031051636, "sampling/importance_sampling_ratio/mean": 0.9958971738815308, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.42790328338742256, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09089764300733805, "clip_ratio/high_max": 0.09089764300733805, "clip_ratio/region_mean": 0.09089764300733805, "reward_total_mean": 0.7246972322463989, "reward_meter_mean": 0.8232419490814209, "reward_meter_std": 0.333120197057724, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9692974090576172, "reward_repeat_soft_std": 0.026825375854969025, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.24479949474334717, "reward_total_composite_mean": 0.7246972322463989, "reward_total_composite_std": 0.298931747674942} {"timestamp_utc": "2026-04-13T05:51:06Z", "mode": "train", "global_step": 2995, "epoch": 0.3008538422903064, "loss": 0.0997, "grad_norm": 7.349178791046143, "learning_rate": 9.272727272727273e-07, "num_tokens": 5604385.0, "completions/mean_length": 70.75, "completions/min_length": 62.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.75, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9736800193786621, "rewards/meter/std": 0.004309943411499262, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8545613288879395, "rewards/repeat_soft/std": 0.08834107220172882, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7967371344566345, "rewards/total_composite/std": 0.024402592331171036, "reward": 0.7967371344566345, "reward_std": 0.024402588605880737, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.042025789618492126, "sampling/sampling_logp_difference/max": 1.1602568626403809, "sampling/importance_sampling_ratio/min": 0.3134056627750397, "sampling/importance_sampling_ratio/mean": 1.0053250789642334, "sampling/importance_sampling_ratio/max": 1.8249489068984985, "entropy": 0.25155870616436005, "clip_ratio/low_mean": 0.016443994361907244, "clip_ratio/low_min": 0.016443994361907244, "clip_ratio/high_mean": 0.02439778856933117, "clip_ratio/high_max": 0.02439778856933117, "clip_ratio/region_mean": 0.04084178293123841, "reward_total_mean": 0.7967371344566345, "reward_meter_mean": 0.9736800193786621, "reward_meter_std": 0.004309943411499262, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8545613288879395, "reward_repeat_soft_std": 0.08834107220172882, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7967371344566345, "reward_total_composite_std": 0.024402592331171036} {"timestamp_utc": "2026-04-13T05:51:13Z", "mode": "train", "global_step": 2996, "epoch": 0.3009542943244601, "loss": 0.0189, "grad_norm": 5.947506904602051, "learning_rate": 9.242424242424244e-07, "num_tokens": 5606434.0, "completions/mean_length": 101.125, "completions/min_length": 96.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.125, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.9910310506820679, "rewards/meter/std": 0.0025674505159258842, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9395633935928345, "rewards/repeat_soft/std": 0.034544624388217926, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.8271703124046326, "rewards/total_composite/std": 0.03518279641866684, "reward": 0.8271703124046326, "reward_std": 0.035182807594537735, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0942489430308342, "sampling/sampling_logp_difference/max": 1.6914716958999634, "sampling/importance_sampling_ratio/min": 0.1842481642961502, "sampling/importance_sampling_ratio/mean": 1.0075531005859375, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4924782030284405, "clip_ratio/low_mean": 0.08265726827085018, "clip_ratio/low_min": 0.08265726827085018, "clip_ratio/high_mean": 0.01608910970389843, "clip_ratio/high_max": 0.01608910970389843, "clip_ratio/region_mean": 0.09874637797474861, "reward_total_mean": 0.8271703124046326, "reward_meter_mean": 0.9910310506820679, "reward_meter_std": 0.0025674505159258842, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9395633935928345, "reward_repeat_soft_std": 0.034544624388217926, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.8271703124046326, "reward_total_composite_std": 0.03518279641866684} {"timestamp_utc": "2026-04-13T05:51:24Z", "mode": "train", "global_step": 2997, "epoch": 0.3010547463586138, "loss": -0.1548, "grad_norm": 2.5151193141937256, "learning_rate": 9.212121212121213e-07, "num_tokens": 5609147.0, "completions/mean_length": 219.125, "completions/min_length": 173.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 177.2857208251953, "completions/min_terminated_length": 173.0, "completions/max_terminated_length": 181.0, "rewards/meter/mean": 0.98818039894104, "rewards/meter/std": 0.0034625448752194643, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7104291915893555, "rewards/repeat_soft/std": 0.05252237617969513, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.23439893126487732, "rewards/total_composite/mean": 0.7977241277694702, "rewards/total_composite/std": 0.06978536397218704, "reward": 0.7977241277694702, "reward_std": 0.06978535652160645, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.056662432849407196, "sampling/sampling_logp_difference/max": 2.0519137382507324, "sampling/importance_sampling_ratio/min": 0.12848877906799316, "sampling/importance_sampling_ratio/mean": 1.0049792528152466, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.19983005337417126, "clip_ratio/low_mean": 0.02818510727956891, "clip_ratio/low_min": 0.02818510727956891, "clip_ratio/high_mean": 0.008542777737602592, "clip_ratio/high_max": 0.008542777737602592, "clip_ratio/region_mean": 0.0367278850171715, "reward_total_mean": 0.7977241277694702, "reward_meter_mean": 0.98818039894104, "reward_meter_std": 0.0034625448752194643, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7104291915893555, "reward_repeat_soft_std": 0.05252237617969513, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.23439893126487732, "reward_total_composite_mean": 0.7977241277694702, "reward_total_composite_std": 0.06978536397218704} {"timestamp_utc": "2026-04-13T05:51:43Z", "mode": "train", "global_step": 2998, "epoch": 0.30115519839276744, "loss": 0.0705, "grad_norm": 21.18613052368164, "learning_rate": 9.181818181818182e-07, "num_tokens": 5610909.0, "completions/mean_length": 66.25, "completions/min_length": 61.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.25, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.8640773892402649, "rewards/meter/std": 0.20965884625911713, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9648085236549377, "rewards/repeat_soft/std": 0.021893059834837914, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.8175656795501709, "rewards/total_composite/std": 0.14315474033355713, "reward": 0.8175656795501709, "reward_std": 0.14315475523471832, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08863905072212219, "sampling/sampling_logp_difference/max": 2.027388334274292, "sampling/importance_sampling_ratio/min": 0.131678968667984, "sampling/importance_sampling_ratio/mean": 0.9817235469818115, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2851592320948839, "clip_ratio/low_mean": 0.027146768872626126, "clip_ratio/low_min": 0.027146768872626126, "clip_ratio/high_mean": 0.021069165552034974, "clip_ratio/high_max": 0.021069165552034974, "clip_ratio/region_mean": 0.0482159344246611, "reward_total_mean": 0.8175656795501709, "reward_meter_mean": 0.8640773892402649, "reward_meter_std": 0.20965884625911713, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9648085236549377, "reward_repeat_soft_std": 0.021893059834837914, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.8175656795501709, "reward_total_composite_std": 0.14315474033355713} {"timestamp_utc": "2026-04-13T05:51:49Z", "mode": "train", "global_step": 2999, "epoch": 0.30125565042692115, "loss": 0.0477, "grad_norm": 8.854698181152344, "learning_rate": 9.151515151515153e-07, "num_tokens": 5612802.0, "completions/mean_length": 61.625, "completions/min_length": 57.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.625, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9861890077590942, "rewards/meter/std": 0.015159180387854576, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9666350483894348, "rewards/repeat_soft/std": 0.02753428742289543, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.8385735750198364, "rewards/total_composite/std": 0.053713977336883545, "reward": 0.8385735750198364, "reward_std": 0.053713951259851456, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10247807204723358, "sampling/sampling_logp_difference/max": 1.487011432647705, "sampling/importance_sampling_ratio/min": 0.22604721784591675, "sampling/importance_sampling_ratio/mean": 0.9974040389060974, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4826372601091862, "clip_ratio/low_mean": 0.05420154589228332, "clip_ratio/low_min": 0.05420154589228332, "clip_ratio/high_mean": 0.010593220591545105, "clip_ratio/high_max": 0.010593220591545105, "clip_ratio/region_mean": 0.06479476648382843, "reward_total_mean": 0.8385735750198364, "reward_meter_mean": 0.9861890077590942, "reward_meter_std": 0.015159180387854576, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9666350483894348, "reward_repeat_soft_std": 0.02753428742289543, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.8385735750198364, "reward_total_composite_std": 0.053713977336883545} {"timestamp_utc": "2026-04-13T05:51:56Z", "mode": "train", "global_step": 3000, "epoch": 0.30135610246107486, "loss": 0.009, "grad_norm": 10.730441093444824, "learning_rate": 9.121212121212122e-07, "num_tokens": 5614582.0, "completions/mean_length": 50.5, "completions/min_length": 46.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.5, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9017293453216553, "rewards/meter/std": 0.10664244741201401, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9844938516616821, "rewards/repeat_soft/std": 0.010284882970154285, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.11310552060604095, "rewards/total_composite/mean": 0.7959775924682617, "rewards/total_composite/std": 0.06556150317192078, "reward": 0.7959775924682617, "reward_std": 0.06556150317192078, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08292357623577118, "sampling/sampling_logp_difference/max": 1.8675518035888672, "sampling/importance_sampling_ratio/min": 0.15450145304203033, "sampling/importance_sampling_ratio/mean": 0.994493305683136, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3449144307523966, "clip_ratio/low_mean": 0.031980258878320456, "clip_ratio/low_min": 0.031980258878320456, "clip_ratio/high_mean": 0.04545543156564236, "clip_ratio/high_max": 0.04545543156564236, "clip_ratio/region_mean": 0.07743569044396281, "reward_total_mean": 0.7959775924682617, "reward_meter_mean": 0.9017293453216553, "reward_meter_std": 0.10664244741201401, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9844938516616821, "reward_repeat_soft_std": 0.010284882970154285, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.11310552060604095, "reward_total_composite_mean": 0.7959775924682617, "reward_total_composite_std": 0.06556150317192078} {"timestamp_utc": "2026-04-13T05:52:49Z", "mode": "eval", "global_step": 3000, "epoch": 0.30135610246107486, "eval_loss": NaN, "eval_runtime": 52.846, "eval_samples_per_second": 1.514, "eval_steps_per_second": 0.189, "eval_num_tokens": 5614582.0, "eval_completions/mean_length": 105.4125, "eval_completions/min_length": 43.3, "eval_completions/max_length": 229.1, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 95.44464416503907, "eval_completions/min_terminated_length": 43.3, "eval_completions/max_terminated_length": 165.8, "eval_rewards/meter/mean": 0.9060098052024841, "eval_rewards/meter/std": 0.18341306671500207, "eval_rewards/count_adherence/mean": 0.9852083325386047, "eval_rewards/count_adherence/std": 0.032573241740465164, "eval_rewards/hard_gate/mean": 0.9625, "eval_rewards/hard_gate/std": 0.10606601536273956, "eval_rewards/repeat_soft/mean": 0.8992622196674347, "eval_rewards/repeat_soft/std": 0.09638850130140782, "eval_rewards/judge_quality/mean": 0.41362500190734863, "eval_rewards/judge_quality/std": 0.13151302188634872, "eval_rewards/total_composite/mean": 0.7464425444602967, "eval_rewards/total_composite/std": 0.14997721426188945, "eval_reward": 0.7464425444602967, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.03655383288860321, "eval_sampling/sampling_logp_difference/max": 0.908556318283081, "eval_sampling/importance_sampling_ratio/min": 0.4131739318370819, "eval_sampling/importance_sampling_ratio/mean": 1.0064593195915221, "eval_sampling/importance_sampling_ratio/max": 1.3055213809013366, "eval_entropy": 0.3632284581661224, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7464425444602967, "eval_reward_meter_mean": 0.9060098052024841, "eval_reward_meter_std": 0.18341306671500207, "eval_reward_count_adherence_mean": 0.9852083325386047, "eval_reward_count_adherence_std": 0.032573241740465164, "eval_reward_hard_gate_mean": 0.9625, "eval_reward_hard_gate_std": 0.10606601536273956, "eval_reward_repeat_soft_mean": 0.8992622196674347, "eval_reward_repeat_soft_std": 0.09638850130140782, "eval_reward_judge_quality_mean": 0.41362500190734863, "eval_reward_judge_quality_std": 0.13151302188634872, "eval_reward_total_composite_mean": 0.7464425444602967, "eval_reward_total_composite_std": 0.14997721426188945} {"timestamp_utc": "2026-04-13T05:53:04Z", "mode": "train", "global_step": 3001, "epoch": 0.3014565544952285, "loss": -0.141, "grad_norm": 1.450937271118164, "learning_rate": 9.090909090909091e-07, "num_tokens": 5616280.0, "completions/mean_length": 107.25, "completions/min_length": 43.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 49.42857360839844, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.8643275499343872, "rewards/meter/std": 0.34899720549583435, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9503395557403564, "rewards/repeat_soft/std": 0.02359302155673504, "rewards/judge_quality/mean": 0.39249998331069946, "rewards/judge_quality/std": 0.13905291259288788, "rewards/total_composite/mean": 0.7190338373184204, "rewards/total_composite/std": 0.29059040546417236, "reward": 0.7190338373184204, "reward_std": 0.29059040546417236, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07556565850973129, "sampling/sampling_logp_difference/max": 1.775634765625, "sampling/importance_sampling_ratio/min": 0.16937589645385742, "sampling/importance_sampling_ratio/mean": 1.003389596939087, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37599456682801247, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08112853579223156, "clip_ratio/high_max": 0.08112853579223156, "clip_ratio/region_mean": 0.08112853579223156, "reward_total_mean": 0.7190338373184204, "reward_meter_mean": 0.8643275499343872, "reward_meter_std": 0.34899720549583435, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9503395557403564, "reward_repeat_soft_std": 0.02359302155673504, "reward_judge_quality_mean": 0.39249998331069946, "reward_judge_quality_std": 0.13905291259288788, "reward_total_composite_mean": 0.7190338373184204, "reward_total_composite_std": 0.29059040546417236} {"timestamp_utc": "2026-04-13T05:53:12Z", "mode": "train", "global_step": 3002, "epoch": 0.3015570065293822, "loss": -0.0087, "grad_norm": 5.752297401428223, "learning_rate": 9.060606060606062e-07, "num_tokens": 5618620.0, "completions/mean_length": 116.5, "completions/min_length": 105.0, "completions/max_length": 127.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.5, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.6411052346229553, "rewards/meter/std": 0.3743896186351776, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7952919006347656, "rewards/repeat_soft/std": 0.05551635101437569, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.6404640674591064, "rewards/total_composite/std": 0.1622922122478485, "reward": 0.6404640674591064, "reward_std": 0.1622922122478485, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06978605687618256, "sampling/sampling_logp_difference/max": 1.756332278251648, "sampling/importance_sampling_ratio/min": 0.17267704010009766, "sampling/importance_sampling_ratio/mean": 0.9937943816184998, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31982115656137466, "clip_ratio/low_mean": 0.017435280606150627, "clip_ratio/low_min": 0.017435280606150627, "clip_ratio/high_mean": 0.04631144739687443, "clip_ratio/high_max": 0.04631144739687443, "clip_ratio/region_mean": 0.06374672800302505, "reward_total_mean": 0.6404640674591064, "reward_meter_mean": 0.6411052346229553, "reward_meter_std": 0.3743896186351776, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7952919006347656, "reward_repeat_soft_std": 0.05551635101437569, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.6404640674591064, "reward_total_composite_std": 0.1622922122478485} {"timestamp_utc": "2026-04-13T05:53:20Z", "mode": "train", "global_step": 3003, "epoch": 0.30165745856353593, "loss": 0.0271, "grad_norm": 9.791135787963867, "learning_rate": 9.030303030303031e-07, "num_tokens": 5620287.0, "completions/mean_length": 61.375, "completions/min_length": 56.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.375, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9885236024856567, "rewards/meter/std": 0.003015915397554636, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9924405813217163, "rewards/repeat_soft/std": 0.008876148611307144, "rewards/judge_quality/mean": 0.6937500238418579, "rewards/judge_quality/std": 0.18554458022117615, "rewards/total_composite/mean": 0.9022046327590942, "rewards/total_composite/std": 0.056359972804784775, "reward": 0.9022046327590942, "reward_std": 0.05635996162891388, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08153125643730164, "sampling/sampling_logp_difference/max": 1.6639080047607422, "sampling/importance_sampling_ratio/min": 0.18939736485481262, "sampling/importance_sampling_ratio/mean": 0.9999184012413025, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43371108174324036, "clip_ratio/low_mean": 0.022600446827709675, "clip_ratio/low_min": 0.022600446827709675, "clip_ratio/high_mean": 0.07207792485132813, "clip_ratio/high_max": 0.07207792485132813, "clip_ratio/region_mean": 0.09467837167903781, "reward_total_mean": 0.9022046327590942, "reward_meter_mean": 0.9885236024856567, "reward_meter_std": 0.003015915397554636, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9924405813217163, "reward_repeat_soft_std": 0.008876148611307144, "reward_judge_quality_mean": 0.6937500238418579, "reward_judge_quality_std": 0.18554458022117615, "reward_total_composite_mean": 0.9022046327590942, "reward_total_composite_std": 0.056359972804784775} {"timestamp_utc": "2026-04-13T05:53:28Z", "mode": "train", "global_step": 3004, "epoch": 0.3017579105976896, "loss": 0.0422, "grad_norm": 9.591343879699707, "learning_rate": 9.000000000000001e-07, "num_tokens": 5621931.0, "completions/mean_length": 59.5, "completions/min_length": 55.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.5, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9772040843963623, "rewards/meter/std": 0.006538757588714361, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9485442638397217, "rewards/repeat_soft/std": 0.05443819984793663, "rewards/judge_quality/mean": 0.5575000047683716, "rewards/judge_quality/std": 0.19955308735370636, "rewards/total_composite/mean": 0.8518462181091309, "rewards/total_composite/std": 0.06158825755119324, "reward": 0.8518462181091309, "reward_std": 0.06158825010061264, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09492659568786621, "sampling/sampling_logp_difference/max": 1.5890369415283203, "sampling/importance_sampling_ratio/min": 0.20412209630012512, "sampling/importance_sampling_ratio/mean": 0.9891846179962158, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.34516322799026966, "clip_ratio/low_mean": 0.04235826199874282, "clip_ratio/low_min": 0.04235826199874282, "clip_ratio/high_mean": 0.023950571659952402, "clip_ratio/high_max": 0.023950571659952402, "clip_ratio/region_mean": 0.06630883365869522, "reward_total_mean": 0.8518462181091309, "reward_meter_mean": 0.9772040843963623, "reward_meter_std": 0.006538757588714361, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9485442638397217, "reward_repeat_soft_std": 0.05443819984793663, "reward_judge_quality_mean": 0.5575000047683716, "reward_judge_quality_std": 0.19955308735370636, "reward_total_composite_mean": 0.8518462181091309, "reward_total_composite_std": 0.06158825755119324} {"timestamp_utc": "2026-04-13T05:53:35Z", "mode": "train", "global_step": 3005, "epoch": 0.3018583626318433, "loss": 0.0211, "grad_norm": 11.146733283996582, "learning_rate": 8.96969696969697e-07, "num_tokens": 5623688.0, "completions/mean_length": 52.625, "completions/min_length": 46.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.625, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9776498675346375, "rewards/meter/std": 0.014370066113770008, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9940676689147949, "rewards/repeat_soft/std": 0.007909293286502361, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404937028885, "rewards/total_composite/mean": 0.8562241792678833, "rewards/total_composite/std": 0.06738311052322388, "reward": 0.8562241792678833, "reward_std": 0.06738311797380447, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09826847165822983, "sampling/sampling_logp_difference/max": 1.6069610118865967, "sampling/importance_sampling_ratio/min": 0.20049600303173065, "sampling/importance_sampling_ratio/mean": 0.9957889318466187, "sampling/importance_sampling_ratio/max": 1.8482433557510376, "entropy": 0.42133957147598267, "clip_ratio/low_mean": 0.05180060304701328, "clip_ratio/low_min": 0.05180060304701328, "clip_ratio/high_mean": 0.02058409620076418, "clip_ratio/high_max": 0.02058409620076418, "clip_ratio/region_mean": 0.07238469924777746, "reward_total_mean": 0.8562241792678833, "reward_meter_mean": 0.9776498675346375, "reward_meter_std": 0.014370066113770008, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9940676689147949, "reward_repeat_soft_std": 0.007909293286502361, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404937028885, "reward_total_composite_mean": 0.8562241792678833, "reward_total_composite_std": 0.06738311052322388} {"timestamp_utc": "2026-04-13T05:53:42Z", "mode": "train", "global_step": 3006, "epoch": 0.301958814665997, "loss": -0.012, "grad_norm": 12.497984886169434, "learning_rate": 8.93939393939394e-07, "num_tokens": 5625108.0, "completions/mean_length": 25.5, "completions/min_length": 23.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.5, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.940778911113739, "rewards/meter/std": 0.05737008899450302, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9519439935684204, "rewards/repeat_soft/std": 0.025683434680104256, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.19334925711154938, "rewards/total_composite/mean": 0.8091699481010437, "rewards/total_composite/std": 0.047165244817733765, "reward": 0.8091699481010437, "reward_std": 0.047165244817733765, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09443464875221252, "sampling/sampling_logp_difference/max": 1.2545523643493652, "sampling/importance_sampling_ratio/min": 0.2852034866809845, "sampling/importance_sampling_ratio/mean": 1.000837802886963, "sampling/importance_sampling_ratio/max": 1.6455843448638916, "entropy": 0.5328217633068562, "clip_ratio/low_mean": 0.040624999441206455, "clip_ratio/low_min": 0.040624999441206455, "clip_ratio/high_mean": 0.056352424435317516, "clip_ratio/high_max": 0.056352424435317516, "clip_ratio/region_mean": 0.09697742387652397, "reward_total_mean": 0.8091699481010437, "reward_meter_mean": 0.940778911113739, "reward_meter_std": 0.05737008899450302, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9519439935684204, "reward_repeat_soft_std": 0.025683434680104256, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.19334925711154938, "reward_total_composite_mean": 0.8091699481010437, "reward_total_composite_std": 0.047165244817733765} {"timestamp_utc": "2026-04-13T05:53:50Z", "mode": "train", "global_step": 3007, "epoch": 0.30205926670015065, "loss": 0.0209, "grad_norm": 6.173469066619873, "learning_rate": 8.90909090909091e-07, "num_tokens": 5627773.0, "completions/mean_length": 143.125, "completions/min_length": 130.0, "completions/max_length": 153.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 143.125, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 153.0, "rewards/meter/mean": 0.992429256439209, "rewards/meter/std": 0.003957211505621672, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9039098620414734, "rewards/repeat_soft/std": 0.05298890545964241, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8129841685295105, "rewards/total_composite/std": 0.00524644972756505, "reward": 0.8129841685295105, "reward_std": 0.005246445070952177, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06465646624565125, "sampling/sampling_logp_difference/max": 1.46826171875, "sampling/importance_sampling_ratio/min": 0.23032550513744354, "sampling/importance_sampling_ratio/mean": 1.0059881210327148, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3356257677078247, "clip_ratio/low_mean": 0.025693088304251432, "clip_ratio/low_min": 0.025693088304251432, "clip_ratio/high_mean": 0.038465611170977354, "clip_ratio/high_max": 0.038465611170977354, "clip_ratio/region_mean": 0.06415869947522879, "reward_total_mean": 0.8129841685295105, "reward_meter_mean": 0.992429256439209, "reward_meter_std": 0.003957211505621672, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9039098620414734, "reward_repeat_soft_std": 0.05298890545964241, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8129841685295105, "reward_total_composite_std": 0.00524644972756505} {"timestamp_utc": "2026-04-13T05:53:58Z", "mode": "train", "global_step": 3008, "epoch": 0.30215971873430436, "loss": 0.0449, "grad_norm": 14.373790740966797, "learning_rate": 8.87878787878788e-07, "num_tokens": 5629495.0, "completions/mean_length": 61.25, "completions/min_length": 58.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.25, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.987136721611023, "rewards/meter/std": 0.008954508230090141, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9549248218536377, "rewards/repeat_soft/std": 0.021624736487865448, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.8104540109634399, "rewards/total_composite/std": 0.017429383471608162, "reward": 0.8104540109634399, "reward_std": 0.01742939092218876, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08406022936105728, "sampling/sampling_logp_difference/max": 1.4222135543823242, "sampling/importance_sampling_ratio/min": 0.24117958545684814, "sampling/importance_sampling_ratio/mean": 0.9962050914764404, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.385408915579319, "clip_ratio/low_mean": 0.01542207831516862, "clip_ratio/low_min": 0.01542207831516862, "clip_ratio/high_mean": 0.06460615177638829, "clip_ratio/high_max": 0.06460615177638829, "clip_ratio/region_mean": 0.0800282300915569, "reward_total_mean": 0.8104540109634399, "reward_meter_mean": 0.987136721611023, "reward_meter_std": 0.008954508230090141, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9549248218536377, "reward_repeat_soft_std": 0.021624736487865448, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.8104540109634399, "reward_total_composite_std": 0.017429383471608162} {"timestamp_utc": "2026-04-13T05:54:07Z", "mode": "train", "global_step": 3009, "epoch": 0.30226017076845807, "loss": 0.0251, "grad_norm": 10.6202974319458, "learning_rate": 8.84848484848485e-07, "num_tokens": 5631656.0, "completions/mean_length": 96.125, "completions/min_length": 90.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.125, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.9815458059310913, "rewards/meter/std": 0.009384870529174805, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9596207141876221, "rewards/repeat_soft/std": 0.010494847781956196, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8136576414108276, "rewards/total_composite/std": 0.0046945419162511826, "reward": 0.8136576414108276, "reward_std": 0.0046945419162511826, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12504608929157257, "sampling/sampling_logp_difference/max": 3.3840084075927734, "sampling/importance_sampling_ratio/min": 0.03391125053167343, "sampling/importance_sampling_ratio/mean": 0.9983288645744324, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4994732700288296, "clip_ratio/low_mean": 0.03675081767141819, "clip_ratio/low_min": 0.03675081767141819, "clip_ratio/high_mean": 0.0767427384853363, "clip_ratio/high_max": 0.0767427384853363, "clip_ratio/region_mean": 0.1134935561567545, "reward_total_mean": 0.8136576414108276, "reward_meter_mean": 0.9815458059310913, "reward_meter_std": 0.009384870529174805, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9596207141876221, "reward_repeat_soft_std": 0.010494847781956196, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8136576414108276, "reward_total_composite_std": 0.0046945419162511826} {"timestamp_utc": "2026-04-13T05:54:14Z", "mode": "train", "global_step": 3010, "epoch": 0.3023606228026118, "loss": 0.017, "grad_norm": 10.867558479309082, "learning_rate": 8.818181818181819e-07, "num_tokens": 5633427.0, "completions/mean_length": 57.375, "completions/min_length": 53.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.375, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.970573902130127, "rewards/meter/std": 0.04786352813243866, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9404629468917847, "rewards/repeat_soft/std": 0.04060623422265053, "rewards/judge_quality/mean": 0.7200000286102295, "rewards/judge_quality/std": 0.20701968669891357, "rewards/total_composite/mean": 0.8968045711517334, "rewards/total_composite/std": 0.05735666677355766, "reward": 0.8968045711517334, "reward_std": 0.057356663048267365, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07066459953784943, "sampling/sampling_logp_difference/max": 1.3194069862365723, "sampling/importance_sampling_ratio/min": 0.2672937512397766, "sampling/importance_sampling_ratio/mean": 0.9993818998336792, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3349027819931507, "clip_ratio/low_mean": 0.03435960691422224, "clip_ratio/low_min": 0.03435960691422224, "clip_ratio/high_mean": 0.0351173032540828, "clip_ratio/high_max": 0.0351173032540828, "clip_ratio/region_mean": 0.06947691016830504, "reward_total_mean": 0.8968045711517334, "reward_meter_mean": 0.970573902130127, "reward_meter_std": 0.04786352813243866, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9404629468917847, "reward_repeat_soft_std": 0.04060623422265053, "reward_judge_quality_mean": 0.7200000286102295, "reward_judge_quality_std": 0.20701968669891357, "reward_total_composite_mean": 0.8968045711517334, "reward_total_composite_std": 0.05735666677355766} {"timestamp_utc": "2026-04-13T05:54:23Z", "mode": "train", "global_step": 3011, "epoch": 0.30246107483676543, "loss": 0.0131, "grad_norm": 3.500725507736206, "learning_rate": 8.787878787878788e-07, "num_tokens": 5636129.0, "completions/mean_length": 140.75, "completions/min_length": 132.0, "completions/max_length": 151.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 140.75, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 151.0, "rewards/meter/mean": 0.9872722625732422, "rewards/meter/std": 0.002930151065811515, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6934366226196289, "rewards/repeat_soft/std": 0.10809318721294403, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.8083661794662476, "rewards/total_composite/std": 0.055452004075050354, "reward": 0.8083661794662476, "reward_std": 0.05545198172330856, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.060675572603940964, "sampling/sampling_logp_difference/max": 2.155946731567383, "sampling/importance_sampling_ratio/min": 0.11579351127147675, "sampling/importance_sampling_ratio/mean": 0.9945259094238281, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.22500311210751534, "clip_ratio/low_mean": 0.035825185710564256, "clip_ratio/low_min": 0.035825185710564256, "clip_ratio/high_mean": 0.009751773439347744, "clip_ratio/high_max": 0.009751773439347744, "clip_ratio/region_mean": 0.045576959149912, "reward_total_mean": 0.8083661794662476, "reward_meter_mean": 0.9872722625732422, "reward_meter_std": 0.002930151065811515, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6934366226196289, "reward_repeat_soft_std": 0.10809318721294403, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.8083661794662476, "reward_total_composite_std": 0.055452004075050354} {"timestamp_utc": "2026-04-13T05:54:30Z", "mode": "train", "global_step": 3012, "epoch": 0.30256152687091914, "loss": 0.0189, "grad_norm": 7.7038254737854, "learning_rate": 8.757575757575758e-07, "num_tokens": 5637916.0, "completions/mean_length": 58.375, "completions/min_length": 55.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.375, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9202978610992432, "rewards/meter/std": 0.12565316259860992, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.948671817779541, "rewards/repeat_soft/std": 0.04753907397389412, "rewards/judge_quality/mean": 0.5612499713897705, "rewards/judge_quality/std": 0.1968638300895691, "rewards/total_composite/mean": 0.8273762464523315, "rewards/total_composite/std": 0.08358938992023468, "reward": 0.8273762464523315, "reward_std": 0.08358937501907349, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0943770706653595, "sampling/sampling_logp_difference/max": 3.4427907466888428, "sampling/importance_sampling_ratio/min": 0.03197532519698143, "sampling/importance_sampling_ratio/mean": 1.0006059408187866, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36210912466049194, "clip_ratio/low_mean": 0.06456281547434628, "clip_ratio/low_min": 0.06456281547434628, "clip_ratio/high_mean": 0.02919910941272974, "clip_ratio/high_max": 0.02919910941272974, "clip_ratio/region_mean": 0.09376192488707602, "reward_total_mean": 0.8273762464523315, "reward_meter_mean": 0.9202978610992432, "reward_meter_std": 0.12565316259860992, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.948671817779541, "reward_repeat_soft_std": 0.04753907397389412, "reward_judge_quality_mean": 0.5612499713897705, "reward_judge_quality_std": 0.1968638300895691, "reward_total_composite_mean": 0.8273762464523315, "reward_total_composite_std": 0.08358938992023468} {"timestamp_utc": "2026-04-13T05:54:38Z", "mode": "train", "global_step": 3013, "epoch": 0.30266197890507285, "loss": 0.0209, "grad_norm": 9.018759727478027, "learning_rate": 8.727272727272728e-07, "num_tokens": 5639694.0, "completions/mean_length": 57.25, "completions/min_length": 53.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.25, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9629620313644409, "rewards/meter/std": 0.035201143473386765, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9821844696998596, "rewards/repeat_soft/std": 0.01242897193878889, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.1865811049938202, "rewards/total_composite/mean": 0.8409263491630554, "rewards/total_composite/std": 0.05773711949586868, "reward": 0.8409263491630554, "reward_std": 0.057737112045288086, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1002926155924797, "sampling/sampling_logp_difference/max": 1.5950334072113037, "sampling/importance_sampling_ratio/min": 0.20290175080299377, "sampling/importance_sampling_ratio/mean": 0.9931233525276184, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39641202986240387, "clip_ratio/low_mean": 0.05421189358457923, "clip_ratio/low_min": 0.05421189358457923, "clip_ratio/high_mean": 0.032858243212103844, "clip_ratio/high_max": 0.032858243212103844, "clip_ratio/region_mean": 0.08707013679668307, "reward_total_mean": 0.8409263491630554, "reward_meter_mean": 0.9629620313644409, "reward_meter_std": 0.035201143473386765, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9821844696998596, "reward_repeat_soft_std": 0.01242897193878889, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.1865811049938202, "reward_total_composite_mean": 0.8409263491630554, "reward_total_composite_std": 0.05773711949586868} {"timestamp_utc": "2026-04-13T05:54:45Z", "mode": "train", "global_step": 3014, "epoch": 0.3027624309392265, "loss": -0.0708, "grad_norm": 9.992512702941895, "learning_rate": 8.696969696969699e-07, "num_tokens": 5641558.0, "completions/mean_length": 59.0, "completions/min_length": 49.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.0, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.77250075340271, "rewards/meter/std": 0.32079440355300903, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8940662741661072, "rewards/repeat_soft/std": 0.09787517786026001, "rewards/judge_quality/mean": 0.35874998569488525, "rewards/judge_quality/std": 0.12205823510885239, "rewards/total_composite/mean": 0.6946569681167603, "rewards/total_composite/std": 0.12384507060050964, "reward": 0.6946569681167603, "reward_std": 0.12384507060050964, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09669715911149979, "sampling/sampling_logp_difference/max": 1.8781137466430664, "sampling/importance_sampling_ratio/min": 0.15287820994853973, "sampling/importance_sampling_ratio/mean": 1.0238968133926392, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5038861706852913, "clip_ratio/low_mean": 0.03629026468843222, "clip_ratio/low_min": 0.03629026468843222, "clip_ratio/high_mean": 0.06144152441993356, "clip_ratio/high_max": 0.06144152441993356, "clip_ratio/region_mean": 0.09773178910836577, "reward_total_mean": 0.6946569681167603, "reward_meter_mean": 0.77250075340271, "reward_meter_std": 0.32079440355300903, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8940662741661072, "reward_repeat_soft_std": 0.09787517786026001, "reward_judge_quality_mean": 0.35874998569488525, "reward_judge_quality_std": 0.12205823510885239, "reward_total_composite_mean": 0.6946569681167603, "reward_total_composite_std": 0.12384507060050964} {"timestamp_utc": "2026-04-13T05:54:52Z", "mode": "train", "global_step": 3015, "epoch": 0.3028628829733802, "loss": 0.0116, "grad_norm": 5.4911909103393555, "learning_rate": 8.666666666666668e-07, "num_tokens": 5643649.0, "completions/mean_length": 65.375, "completions/min_length": 62.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.375, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9880605936050415, "rewards/meter/std": 0.002129737753421068, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7643797993659973, "rewards/repeat_soft/std": 0.051391951739788055, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7906902432441711, "rewards/total_composite/std": 0.0160442553460598, "reward": 0.7906902432441711, "reward_std": 0.01604425720870495, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06432919949293137, "sampling/sampling_logp_difference/max": 1.6077264547348022, "sampling/importance_sampling_ratio/min": 0.2003425806760788, "sampling/importance_sampling_ratio/mean": 1.0024696588516235, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.25780429877340794, "clip_ratio/low_mean": 0.009469697251915932, "clip_ratio/low_min": 0.009469697251915932, "clip_ratio/high_mean": 0.057833022903651, "clip_ratio/high_max": 0.057833022903651, "clip_ratio/region_mean": 0.06730272015556693, "reward_total_mean": 0.7906902432441711, "reward_meter_mean": 0.9880605936050415, "reward_meter_std": 0.002129737753421068, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7643797993659973, "reward_repeat_soft_std": 0.051391951739788055, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7906902432441711, "reward_total_composite_std": 0.0160442553460598} {"timestamp_utc": "2026-04-13T05:55:00Z", "mode": "train", "global_step": 3016, "epoch": 0.3029633350075339, "loss": -0.0238, "grad_norm": 10.439040184020996, "learning_rate": 8.636363636363637e-07, "num_tokens": 5645387.0, "completions/mean_length": 56.25, "completions/min_length": 50.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.25, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9460763931274414, "rewards/meter/std": 0.08218027651309967, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.986307680606842, "rewards/repeat_soft/std": 0.017491811886429787, "rewards/judge_quality/mean": 0.8575000166893005, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.9316151142120361, "rewards/total_composite/std": 0.05789894238114357, "reward": 0.9316151142120361, "reward_std": 0.05789892002940178, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09170816093683243, "sampling/sampling_logp_difference/max": 1.4005790948867798, "sampling/importance_sampling_ratio/min": 0.24645420908927917, "sampling/importance_sampling_ratio/mean": 0.9964372515678406, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4069794528186321, "clip_ratio/low_mean": 0.007129629608243704, "clip_ratio/low_min": 0.007129629608243704, "clip_ratio/high_mean": 0.07826740760356188, "clip_ratio/high_max": 0.07826740760356188, "clip_ratio/region_mean": 0.08539703721180558, "reward_total_mean": 0.9316151142120361, "reward_meter_mean": 0.9460763931274414, "reward_meter_std": 0.08218027651309967, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.986307680606842, "reward_repeat_soft_std": 0.017491811886429787, "reward_judge_quality_mean": 0.8575000166893005, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.9316151142120361, "reward_total_composite_std": 0.05789894238114357} {"timestamp_utc": "2026-04-13T05:55:07Z", "mode": "train", "global_step": 3017, "epoch": 0.30306378704168757, "loss": 0.0055, "grad_norm": 5.790579319000244, "learning_rate": 8.606060606060607e-07, "num_tokens": 5648108.0, "completions/mean_length": 159.125, "completions/min_length": 148.0, "completions/max_length": 165.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 159.125, "completions/min_terminated_length": 148.0, "completions/max_terminated_length": 165.0, "rewards/meter/mean": 0.9531643390655518, "rewards/meter/std": 0.04967638477683067, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8871544599533081, "rewards/repeat_soft/std": 0.05077657848596573, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7936394214630127, "rewards/total_composite/std": 0.024485090747475624, "reward": 0.7936394214630127, "reward_std": 0.024485094472765923, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07696676254272461, "sampling/sampling_logp_difference/max": 2.4683568477630615, "sampling/importance_sampling_ratio/min": 0.08472395688295364, "sampling/importance_sampling_ratio/mean": 1.0036821365356445, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.32174297980964184, "clip_ratio/low_mean": 0.008969807997345924, "clip_ratio/low_min": 0.008969807997345924, "clip_ratio/high_mean": 0.0443093478679657, "clip_ratio/high_max": 0.0443093478679657, "clip_ratio/region_mean": 0.05327915586531162, "reward_total_mean": 0.7936394214630127, "reward_meter_mean": 0.9531643390655518, "reward_meter_std": 0.04967638477683067, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8871544599533081, "reward_repeat_soft_std": 0.05077657848596573, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7936394214630127, "reward_total_composite_std": 0.024485090747475624} {"timestamp_utc": "2026-04-13T05:55:19Z", "mode": "train", "global_step": 3018, "epoch": 0.3031642390758413, "loss": -0.1298, "grad_norm": 1.2859692573547363, "learning_rate": 8.575757575757576e-07, "num_tokens": 5649759.0, "completions/mean_length": 102.375, "completions/min_length": 41.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 43.85714340209961, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9528570175170898, "rewards/meter/std": 0.04140763729810715, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9476414918899536, "rewards/repeat_soft/std": 0.08326180279254913, "rewards/judge_quality/mean": 0.41875001788139343, "rewards/judge_quality/std": 0.18074746429920197, "rewards/total_composite/mean": 0.718055248260498, "rewards/total_composite/std": 0.2913033366203308, "reward": 0.718055248260498, "reward_std": 0.2913033664226532, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.054273322224617004, "sampling/sampling_logp_difference/max": 0.8370587825775146, "sampling/importance_sampling_ratio/min": 0.4571608901023865, "sampling/importance_sampling_ratio/mean": 1.0167490243911743, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.26442351937294006, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.04607227258384228, "clip_ratio/high_max": 0.04607227258384228, "clip_ratio/region_mean": 0.04607227258384228, "reward_total_mean": 0.718055248260498, "reward_meter_mean": 0.9528570175170898, "reward_meter_std": 0.04140763729810715, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9476414918899536, "reward_repeat_soft_std": 0.08326180279254913, "reward_judge_quality_mean": 0.41875001788139343, "reward_judge_quality_std": 0.18074746429920197, "reward_total_composite_mean": 0.718055248260498, "reward_total_composite_std": 0.2913033366203308} {"timestamp_utc": "2026-04-13T05:55:26Z", "mode": "train", "global_step": 3019, "epoch": 0.303264691109995, "loss": 0.022, "grad_norm": 12.010717391967773, "learning_rate": 8.545454545454546e-07, "num_tokens": 5651738.0, "completions/mean_length": 73.375, "completions/min_length": 67.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.375, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.8301541805267334, "rewards/meter/std": 0.26943501830101013, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9565457105636597, "rewards/repeat_soft/std": 0.036038175225257874, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.7575989365577698, "rewards/total_composite/std": 0.07677193731069565, "reward": 0.7575989365577698, "reward_std": 0.07677192986011505, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11713699251413345, "sampling/sampling_logp_difference/max": 1.9868943691253662, "sampling/importance_sampling_ratio/min": 0.13712060451507568, "sampling/importance_sampling_ratio/mean": 1.0042226314544678, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5213466957211494, "clip_ratio/low_mean": 0.03105978248640895, "clip_ratio/low_min": 0.03105978248640895, "clip_ratio/high_mean": 0.08601309824734926, "clip_ratio/high_max": 0.08601309824734926, "clip_ratio/region_mean": 0.11707288073375821, "reward_total_mean": 0.7575989365577698, "reward_meter_mean": 0.8301541805267334, "reward_meter_std": 0.26943501830101013, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9565457105636597, "reward_repeat_soft_std": 0.036038175225257874, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.7575989365577698, "reward_total_composite_std": 0.07677193731069565} {"timestamp_utc": "2026-04-13T05:55:33Z", "mode": "train", "global_step": 3020, "epoch": 0.3033651431441487, "loss": 0.036, "grad_norm": 8.453275680541992, "learning_rate": 8.515151515151515e-07, "num_tokens": 5654194.0, "completions/mean_length": 123.0, "completions/min_length": 105.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 123.0, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.96962571144104, "rewards/meter/std": 0.05716070532798767, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8505904078483582, "rewards/repeat_soft/std": 0.04946720972657204, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7872655987739563, "rewards/total_composite/std": 0.03683427721261978, "reward": 0.7872655987739563, "reward_std": 0.03683427721261978, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07700465619564056, "sampling/sampling_logp_difference/max": 1.861710786819458, "sampling/importance_sampling_ratio/min": 0.15540653467178345, "sampling/importance_sampling_ratio/mean": 1.0022157430648804, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3865550830960274, "clip_ratio/low_mean": 0.017821429297327995, "clip_ratio/low_min": 0.017821429297327995, "clip_ratio/high_mean": 0.05101762805134058, "clip_ratio/high_max": 0.05101762805134058, "clip_ratio/region_mean": 0.06883905734866858, "reward_total_mean": 0.7872655987739563, "reward_meter_mean": 0.96962571144104, "reward_meter_std": 0.05716070532798767, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8505904078483582, "reward_repeat_soft_std": 0.04946720972657204, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7872655987739563, "reward_total_composite_std": 0.03683427721261978} {"timestamp_utc": "2026-04-13T05:55:39Z", "mode": "train", "global_step": 3021, "epoch": 0.30346559517830235, "loss": -0.0281, "grad_norm": 7.13665771484375, "learning_rate": 8.484848484848486e-07, "num_tokens": 5655707.0, "completions/mean_length": 34.125, "completions/min_length": 30.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.125, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.9914454221725464, "rewards/meter/std": 0.0037945141084492207, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9359663724899292, "rewards/repeat_soft/std": 0.04727697744965553, "rewards/judge_quality/mean": 0.38874998688697815, "rewards/judge_quality/std": 0.08675704896450043, "rewards/total_composite/mean": 0.8063720464706421, "rewards/total_composite/std": 0.02680456079542637, "reward": 0.8063720464706421, "reward_std": 0.026804562658071518, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07900074124336243, "sampling/sampling_logp_difference/max": 0.7491797208786011, "sampling/importance_sampling_ratio/min": 0.47275421023368835, "sampling/importance_sampling_ratio/mean": 1.0162237882614136, "sampling/importance_sampling_ratio/max": 1.8248097896575928, "entropy": 0.4744841121137142, "clip_ratio/low_mean": 0.042220113798975945, "clip_ratio/low_min": 0.042220113798975945, "clip_ratio/high_mean": 0.07754298159852624, "clip_ratio/high_max": 0.07754298159852624, "clip_ratio/region_mean": 0.11976309539750218, "reward_total_mean": 0.8063720464706421, "reward_meter_mean": 0.9914454221725464, "reward_meter_std": 0.0037945141084492207, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9359663724899292, "reward_repeat_soft_std": 0.04727697744965553, "reward_judge_quality_mean": 0.38874998688697815, "reward_judge_quality_std": 0.08675704896450043, "reward_total_composite_mean": 0.8063720464706421, "reward_total_composite_std": 0.02680456079542637} {"timestamp_utc": "2026-04-13T05:55:45Z", "mode": "train", "global_step": 3022, "epoch": 0.30356604721245606, "loss": 0.0182, "grad_norm": 26.677947998046875, "learning_rate": 8.454545454545456e-07, "num_tokens": 5657487.0, "completions/mean_length": 55.5, "completions/min_length": 49.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.5, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.7592220902442932, "rewards/meter/std": 0.3668665885925293, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.972699761390686, "rewards/repeat_soft/std": 0.016019387170672417, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.7370449304580688, "rewards/total_composite/std": 0.1877281814813614, "reward": 0.7370449304580688, "reward_std": 0.1877281814813614, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12219379842281342, "sampling/sampling_logp_difference/max": 2.532571792602539, "sampling/importance_sampling_ratio/min": 0.07945441454648972, "sampling/importance_sampling_ratio/mean": 0.997035562992096, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4702206030488014, "clip_ratio/low_mean": 0.027099364437162876, "clip_ratio/low_min": 0.027099364437162876, "clip_ratio/high_mean": 0.09456007368862629, "clip_ratio/high_max": 0.09456007368862629, "clip_ratio/region_mean": 0.12165943812578917, "reward_total_mean": 0.7370449304580688, "reward_meter_mean": 0.7592220902442932, "reward_meter_std": 0.3668665885925293, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.972699761390686, "reward_repeat_soft_std": 0.016019387170672417, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.7370449304580688, "reward_total_composite_std": 0.1877281814813614} {"timestamp_utc": "2026-04-13T05:55:51Z", "mode": "train", "global_step": 3023, "epoch": 0.30366649924660977, "loss": 0.0226, "grad_norm": 9.811635971069336, "learning_rate": 8.424242424242425e-07, "num_tokens": 5659282.0, "completions/mean_length": 71.375, "completions/min_length": 68.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.375, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9870806932449341, "rewards/meter/std": 0.0034849653020501137, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.961127758026123, "rewards/repeat_soft/std": 0.05019128695130348, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.1249857097864151, "rewards/total_composite/mean": 0.7960491180419922, "rewards/total_composite/std": 0.04090190306305885, "reward": 0.7960491180419922, "reward_std": 0.04090190306305885, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10556749999523163, "sampling/sampling_logp_difference/max": 2.05064058303833, "sampling/importance_sampling_ratio/min": 0.12865246832370758, "sampling/importance_sampling_ratio/mean": 1.0031136274337769, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.428768340498209, "clip_ratio/low_mean": 0.02136171516031027, "clip_ratio/low_min": 0.02136171516031027, "clip_ratio/high_mean": 0.07008374761790037, "clip_ratio/high_max": 0.07008374761790037, "clip_ratio/region_mean": 0.09144546277821064, "reward_total_mean": 0.7960491180419922, "reward_meter_mean": 0.9870806932449341, "reward_meter_std": 0.0034849653020501137, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.961127758026123, "reward_repeat_soft_std": 0.05019128695130348, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.1249857097864151, "reward_total_composite_mean": 0.7960491180419922, "reward_total_composite_std": 0.04090190306305885} {"timestamp_utc": "2026-04-13T05:55:59Z", "mode": "train", "global_step": 3024, "epoch": 0.3037669512807634, "loss": 0.0382, "grad_norm": 8.732715606689453, "learning_rate": 8.393939393939395e-07, "num_tokens": 5661621.0, "completions/mean_length": 114.375, "completions/min_length": 103.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.375, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.8864529132843018, "rewards/meter/std": 0.23007607460021973, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.987713098526001, "rewards/repeat_soft/std": 0.008916349150240421, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.7961751222610474, "rewards/total_composite/std": 0.11853066086769104, "reward": 0.7961751222610474, "reward_std": 0.11853066086769104, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1074879840016365, "sampling/sampling_logp_difference/max": 2.7770133018493652, "sampling/importance_sampling_ratio/min": 0.06222407519817352, "sampling/importance_sampling_ratio/mean": 1.0046595335006714, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4200197160243988, "clip_ratio/low_mean": 0.012820512987673283, "clip_ratio/low_min": 0.012820512987673283, "clip_ratio/high_mean": 0.09251427045091987, "clip_ratio/high_max": 0.09251427045091987, "clip_ratio/region_mean": 0.10533478343859315, "reward_total_mean": 0.7961751222610474, "reward_meter_mean": 0.8864529132843018, "reward_meter_std": 0.23007607460021973, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.987713098526001, "reward_repeat_soft_std": 0.008916349150240421, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.7961751222610474, "reward_total_composite_std": 0.11853066086769104} {"timestamp_utc": "2026-04-13T05:56:05Z", "mode": "train", "global_step": 3025, "epoch": 0.30386740331491713, "loss": 0.0344, "grad_norm": 10.661894798278809, "learning_rate": 8.363636363636364e-07, "num_tokens": 5663131.0, "completions/mean_length": 42.75, "completions/min_length": 40.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.75, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.973373532295227, "rewards/meter/std": 0.012187262065708637, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9487019181251526, "rewards/repeat_soft/std": 0.03915436938405037, "rewards/judge_quality/mean": 0.48374998569488525, "rewards/judge_quality/std": 0.10809222608804703, "rewards/total_composite/mean": 0.8280132412910461, "rewards/total_composite/std": 0.030130572617053986, "reward": 0.8280132412910461, "reward_std": 0.03013056516647339, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09914009273052216, "sampling/sampling_logp_difference/max": 1.4486351013183594, "sampling/importance_sampling_ratio/min": 0.23489068448543549, "sampling/importance_sampling_ratio/mean": 1.0084238052368164, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5724952295422554, "clip_ratio/low_mean": 0.08658943697810173, "clip_ratio/low_min": 0.08658943697810173, "clip_ratio/high_mean": 0.015243902802467346, "clip_ratio/high_max": 0.015243902802467346, "clip_ratio/region_mean": 0.10183333978056908, "reward_total_mean": 0.8280132412910461, "reward_meter_mean": 0.973373532295227, "reward_meter_std": 0.012187262065708637, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9487019181251526, "reward_repeat_soft_std": 0.03915436938405037, "reward_judge_quality_mean": 0.48374998569488525, "reward_judge_quality_std": 0.10809222608804703, "reward_total_composite_mean": 0.8280132412910461, "reward_total_composite_std": 0.030130572617053986} {"timestamp_utc": "2026-04-13T05:56:12Z", "mode": "train", "global_step": 3026, "epoch": 0.30396785534907084, "loss": 0.0422, "grad_norm": 9.278218269348145, "learning_rate": 8.333333333333333e-07, "num_tokens": 5664898.0, "completions/mean_length": 64.875, "completions/min_length": 59.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.875, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9941222667694092, "rewards/meter/std": 0.0033438042737543583, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9767820835113525, "rewards/repeat_soft/std": 0.019111772999167442, "rewards/judge_quality/mean": 0.7262500524520874, "rewards/judge_quality/std": 0.19639156758785248, "rewards/total_composite/mean": 0.9129081964492798, "rewards/total_composite/std": 0.05913781374692917, "reward": 0.9129081964492798, "reward_std": 0.05913781747221947, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09383337199687958, "sampling/sampling_logp_difference/max": 2.001037359237671, "sampling/importance_sampling_ratio/min": 0.13519497215747833, "sampling/importance_sampling_ratio/mean": 1.014851450920105, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43895626813173294, "clip_ratio/low_mean": 0.028181818313896656, "clip_ratio/low_min": 0.028181818313896656, "clip_ratio/high_mean": 0.050866622012108564, "clip_ratio/high_max": 0.050866622012108564, "clip_ratio/region_mean": 0.07904844032600522, "reward_total_mean": 0.9129081964492798, "reward_meter_mean": 0.9941222667694092, "reward_meter_std": 0.0033438042737543583, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9767820835113525, "reward_repeat_soft_std": 0.019111772999167442, "reward_judge_quality_mean": 0.7262500524520874, "reward_judge_quality_std": 0.19639156758785248, "reward_total_composite_mean": 0.9129081964492798, "reward_total_composite_std": 0.05913781374692917} {"timestamp_utc": "2026-04-13T05:56:18Z", "mode": "train", "global_step": 3027, "epoch": 0.3040683073832245, "loss": 0.012, "grad_norm": 11.309938430786133, "learning_rate": 8.303030303030303e-07, "num_tokens": 5666510.0, "completions/mean_length": 38.5, "completions/min_length": 35.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.5, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9934875965118408, "rewards/meter/std": 0.0026954165659844875, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9372363090515137, "rewards/repeat_soft/std": 0.03513173386454582, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.19255799055099487, "rewards/total_composite/mean": 0.8325430750846863, "rewards/total_composite/std": 0.059845153242349625, "reward": 0.8325430750846863, "reward_std": 0.059845153242349625, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09215793013572693, "sampling/sampling_logp_difference/max": 1.5811407566070557, "sampling/importance_sampling_ratio/min": 0.20574025809764862, "sampling/importance_sampling_ratio/mean": 1.0197697877883911, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4628109000623226, "clip_ratio/low_mean": 0.0679437043145299, "clip_ratio/low_min": 0.0679437043145299, "clip_ratio/high_mean": 0.009868420660495758, "clip_ratio/high_max": 0.009868420660495758, "clip_ratio/region_mean": 0.07781212497502565, "reward_total_mean": 0.8325430750846863, "reward_meter_mean": 0.9934875965118408, "reward_meter_std": 0.0026954165659844875, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9372363090515137, "reward_repeat_soft_std": 0.03513173386454582, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.19255799055099487, "reward_total_composite_mean": 0.8325430750846863, "reward_total_composite_std": 0.059845153242349625} {"timestamp_utc": "2026-04-13T05:56:25Z", "mode": "train", "global_step": 3028, "epoch": 0.3041687594173782, "loss": 0.0252, "grad_norm": 5.901096343994141, "learning_rate": 8.272727272727274e-07, "num_tokens": 5668845.0, "completions/mean_length": 107.875, "completions/min_length": 97.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.875, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.9895512461662292, "rewards/meter/std": 0.004683605395257473, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8589791655540466, "rewards/repeat_soft/std": 0.08936511725187302, "rewards/judge_quality/mean": 0.3137499690055847, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7753209471702576, "rewards/total_composite/std": 0.03178638219833374, "reward": 0.7753209471702576, "reward_std": 0.03178637474775314, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07308775186538696, "sampling/sampling_logp_difference/max": 1.6328158378601074, "sampling/importance_sampling_ratio/min": 0.1953786462545395, "sampling/importance_sampling_ratio/mean": 1.0086742639541626, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3464006036520004, "clip_ratio/low_mean": 0.03822151804342866, "clip_ratio/low_min": 0.03822151804342866, "clip_ratio/high_mean": 0.035149941220879555, "clip_ratio/high_max": 0.035149941220879555, "clip_ratio/region_mean": 0.07337145926430821, "reward_total_mean": 0.7753209471702576, "reward_meter_mean": 0.9895512461662292, "reward_meter_std": 0.004683605395257473, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8589791655540466, "reward_repeat_soft_std": 0.08936511725187302, "reward_judge_quality_mean": 0.3137499690055847, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7753209471702576, "reward_total_composite_std": 0.03178638219833374} {"timestamp_utc": "2026-04-13T05:56:32Z", "mode": "train", "global_step": 3029, "epoch": 0.3042692114515319, "loss": 0.0145, "grad_norm": 3.1686794757843018, "learning_rate": 8.242424242424244e-07, "num_tokens": 5671152.0, "completions/mean_length": 113.375, "completions/min_length": 111.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.375, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.9735478758811951, "rewards/meter/std": 0.008759287185966969, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6200897693634033, "rewards/repeat_soft/std": 0.0550968199968338, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.12351980060338974, "rewards/total_composite/mean": 0.7431055307388306, "rewards/total_composite/std": 0.03883467987179756, "reward": 0.7431055307388306, "reward_std": 0.038834672421216965, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03230461850762367, "sampling/sampling_logp_difference/max": 1.530694603919983, "sampling/importance_sampling_ratio/min": 0.21638531982898712, "sampling/importance_sampling_ratio/mean": 1.005801796913147, "sampling/importance_sampling_ratio/max": 1.9802637100219727, "entropy": 0.1539660757407546, "clip_ratio/low_mean": 0.013160266680642962, "clip_ratio/low_min": 0.013160266680642962, "clip_ratio/high_mean": 0.015558321494609118, "clip_ratio/high_max": 0.015558321494609118, "clip_ratio/region_mean": 0.02871858817525208, "reward_total_mean": 0.7431055307388306, "reward_meter_mean": 0.9735478758811951, "reward_meter_std": 0.008759287185966969, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6200897693634033, "reward_repeat_soft_std": 0.0550968199968338, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.12351980060338974, "reward_total_composite_mean": 0.7431055307388306, "reward_total_composite_std": 0.03883467987179756} {"timestamp_utc": "2026-04-13T05:56:38Z", "mode": "train", "global_step": 3030, "epoch": 0.30436966348568556, "loss": -0.0059, "grad_norm": 7.8243794441223145, "learning_rate": 8.212121212121213e-07, "num_tokens": 5672826.0, "completions/mean_length": 58.25, "completions/min_length": 54.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.25, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9487978219985962, "rewards/meter/std": 0.07016287744045258, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9520066976547241, "rewards/repeat_soft/std": 0.054188333451747894, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.8169097304344177, "rewards/total_composite/std": 0.0696154311299324, "reward": 0.8169097304344177, "reward_std": 0.0696154311299324, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08044353872537613, "sampling/sampling_logp_difference/max": 1.806551218032837, "sampling/importance_sampling_ratio/min": 0.16421952843666077, "sampling/importance_sampling_ratio/mean": 0.9881746768951416, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.33009860664606094, "clip_ratio/low_mean": 0.02822718769311905, "clip_ratio/low_min": 0.02822718769311905, "clip_ratio/high_mean": 0.0363581373821944, "clip_ratio/high_max": 0.0363581373821944, "clip_ratio/region_mean": 0.06458532507531345, "reward_total_mean": 0.8169097304344177, "reward_meter_mean": 0.9487978219985962, "reward_meter_std": 0.07016287744045258, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9520066976547241, "reward_repeat_soft_std": 0.054188333451747894, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.8169097304344177, "reward_total_composite_std": 0.0696154311299324} {"timestamp_utc": "2026-04-13T05:56:45Z", "mode": "train", "global_step": 3031, "epoch": 0.30447011551983927, "loss": 0.0466, "grad_norm": 10.539841651916504, "learning_rate": 8.181818181818182e-07, "num_tokens": 5674531.0, "completions/mean_length": 57.125, "completions/min_length": 54.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.125, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9629682302474976, "rewards/meter/std": 0.03803578019142151, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9574309587478638, "rewards/repeat_soft/std": 0.09213971346616745, "rewards/judge_quality/mean": 0.6487500071525574, "rewards/judge_quality/std": 0.2456151396036148, "rewards/total_composite/mean": 0.8737037181854248, "rewards/total_composite/std": 0.07920722663402557, "reward": 0.8737037181854248, "reward_std": 0.07920721173286438, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10213761776685715, "sampling/sampling_logp_difference/max": 3.625711441040039, "sampling/importance_sampling_ratio/min": 0.02663014456629753, "sampling/importance_sampling_ratio/mean": 1.002992033958435, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39099763333797455, "clip_ratio/low_mean": 0.04411326488479972, "clip_ratio/low_min": 0.04411326488479972, "clip_ratio/high_mean": 0.041829004883766174, "clip_ratio/high_max": 0.041829004883766174, "clip_ratio/region_mean": 0.0859422697685659, "reward_total_mean": 0.8737037181854248, "reward_meter_mean": 0.9629682302474976, "reward_meter_std": 0.03803578019142151, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9574309587478638, "reward_repeat_soft_std": 0.09213971346616745, "reward_judge_quality_mean": 0.6487500071525574, "reward_judge_quality_std": 0.2456151396036148, "reward_total_composite_mean": 0.8737037181854248, "reward_total_composite_std": 0.07920722663402557} {"timestamp_utc": "2026-04-13T05:56:52Z", "mode": "train", "global_step": 3032, "epoch": 0.304570567553993, "loss": 0.007, "grad_norm": 10.10116195678711, "learning_rate": 8.151515151515152e-07, "num_tokens": 5676545.0, "completions/mean_length": 84.75, "completions/min_length": 79.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.75, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.6496614813804626, "rewards/meter/std": 0.3749837577342987, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9459519386291504, "rewards/repeat_soft/std": 0.03996838629245758, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.6565678715705872, "rewards/total_composite/std": 0.16052785515785217, "reward": 0.6565678715705872, "reward_std": 0.16052788496017456, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1053750142455101, "sampling/sampling_logp_difference/max": 2.346013307571411, "sampling/importance_sampling_ratio/min": 0.09575013071298599, "sampling/importance_sampling_ratio/mean": 0.9928188920021057, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.41013891249895096, "clip_ratio/low_mean": 0.05460651870816946, "clip_ratio/low_min": 0.05460651870816946, "clip_ratio/high_mean": 0.04563112370669842, "clip_ratio/high_max": 0.04563112370669842, "clip_ratio/region_mean": 0.10023764241486788, "reward_total_mean": 0.6565678715705872, "reward_meter_mean": 0.6496614813804626, "reward_meter_std": 0.3749837577342987, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9459519386291504, "reward_repeat_soft_std": 0.03996838629245758, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.6565678715705872, "reward_total_composite_std": 0.16052785515785217} {"timestamp_utc": "2026-04-13T05:56:58Z", "mode": "train", "global_step": 3033, "epoch": 0.3046710195881467, "loss": 0.0789, "grad_norm": 9.140888214111328, "learning_rate": 8.121212121212121e-07, "num_tokens": 5677866.0, "completions/mean_length": 28.125, "completions/min_length": 23.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.125, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.9821134805679321, "rewards/meter/std": 0.023182131350040436, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9547449350357056, "rewards/repeat_soft/std": 0.015329758636653423, "rewards/judge_quality/mean": 0.5049999952316284, "rewards/judge_quality/std": 0.16801361739635468, "rewards/total_composite/mean": 0.8389256000518799, "rewards/total_composite/std": 0.04801037535071373, "reward": 0.8389256000518799, "reward_std": 0.04801037535071373, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06317402422428131, "sampling/sampling_logp_difference/max": 1.4559412002563477, "sampling/importance_sampling_ratio/min": 0.23318080604076385, "sampling/importance_sampling_ratio/mean": 1.0033586025238037, "sampling/importance_sampling_ratio/max": 1.6073142290115356, "entropy": 0.31557659804821014, "clip_ratio/low_mean": 0.051790451630949974, "clip_ratio/low_min": 0.051790451630949974, "clip_ratio/high_mean": 0.010869565419852734, "clip_ratio/high_max": 0.010869565419852734, "clip_ratio/region_mean": 0.06266001705080271, "reward_total_mean": 0.8389256000518799, "reward_meter_mean": 0.9821134805679321, "reward_meter_std": 0.023182131350040436, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9547449350357056, "reward_repeat_soft_std": 0.015329758636653423, "reward_judge_quality_mean": 0.5049999952316284, "reward_judge_quality_std": 0.16801361739635468, "reward_total_composite_mean": 0.8389256000518799, "reward_total_composite_std": 0.04801037535071373} {"timestamp_utc": "2026-04-13T05:57:10Z", "mode": "train", "global_step": 3034, "epoch": 0.30477147162230034, "loss": -0.1065, "grad_norm": 2.1272363662719727, "learning_rate": 8.09090909090909e-07, "num_tokens": 5679354.0, "completions/mean_length": 96.0, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 36.57143020629883, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.7720808982849121, "rewards/meter/std": 0.2199525535106659, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9746736884117126, "rewards/repeat_soft/std": 0.03248816356062889, "rewards/judge_quality/mean": 0.5649999976158142, "rewards/judge_quality/std": 0.3206244111061096, "rewards/total_composite/mean": 0.7084578275680542, "rewards/total_composite/std": 0.2985589802265167, "reward": 0.7084578275680542, "reward_std": 0.2985589802265167, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10667513310909271, "sampling/sampling_logp_difference/max": 1.6401805877685547, "sampling/importance_sampling_ratio/min": 0.1939450204372406, "sampling/importance_sampling_ratio/mean": 0.996523916721344, "sampling/importance_sampling_ratio/max": 1.8939929008483887, "entropy": 0.4288243390619755, "clip_ratio/low_mean": 0.01689189113676548, "clip_ratio/low_min": 0.01689189113676548, "clip_ratio/high_mean": 0.0754359238781035, "clip_ratio/high_max": 0.0754359238781035, "clip_ratio/region_mean": 0.09232781501486897, "reward_total_mean": 0.7084578275680542, "reward_meter_mean": 0.7720808982849121, "reward_meter_std": 0.2199525535106659, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9746736884117126, "reward_repeat_soft_std": 0.03248816356062889, "reward_judge_quality_mean": 0.5649999976158142, "reward_judge_quality_std": 0.3206244111061096, "reward_total_composite_mean": 0.7084578275680542, "reward_total_composite_std": 0.2985589802265167} {"timestamp_utc": "2026-04-13T05:57:18Z", "mode": "train", "global_step": 3035, "epoch": 0.30487192365645405, "loss": 0.0023, "grad_norm": 5.293359756469727, "learning_rate": 8.060606060606062e-07, "num_tokens": 5681700.0, "completions/mean_length": 131.25, "completions/min_length": 102.0, "completions/max_length": 142.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.25, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 142.0, "rewards/meter/mean": 0.9894780516624451, "rewards/meter/std": 0.0038501506205648184, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8260504603385925, "rewards/repeat_soft/std": 0.06089615076780319, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.8366826772689819, "rewards/total_composite/std": 0.07326877117156982, "reward": 0.8366826772689819, "reward_std": 0.07326877862215042, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08031944185495377, "sampling/sampling_logp_difference/max": 2.981020927429199, "sampling/importance_sampling_ratio/min": 0.050741005688905716, "sampling/importance_sampling_ratio/mean": 1.0104175806045532, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3710548337548971, "clip_ratio/low_mean": 0.053227005526423454, "clip_ratio/low_min": 0.053227005526423454, "clip_ratio/high_mean": 0.008528660982847214, "clip_ratio/high_max": 0.008528660982847214, "clip_ratio/region_mean": 0.06175566650927067, "reward_total_mean": 0.8366826772689819, "reward_meter_mean": 0.9894780516624451, "reward_meter_std": 0.0038501506205648184, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8260504603385925, "reward_repeat_soft_std": 0.06089615076780319, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.8366826772689819, "reward_total_composite_std": 0.07326877117156982} {"timestamp_utc": "2026-04-13T05:57:25Z", "mode": "train", "global_step": 3036, "epoch": 0.30497237569060776, "loss": -0.0044, "grad_norm": 12.38042163848877, "learning_rate": 8.030303030303031e-07, "num_tokens": 5683037.0, "completions/mean_length": 32.125, "completions/min_length": 29.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.125, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.989794135093689, "rewards/meter/std": 0.00534153962507844, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9452117085456848, "rewards/repeat_soft/std": 0.03241921588778496, "rewards/judge_quality/mean": 0.39249998331069946, "rewards/judge_quality/std": 0.08892211318016052, "rewards/total_composite/mean": 0.8076785206794739, "rewards/total_composite/std": 0.025669461116194725, "reward": 0.8076785206794739, "reward_std": 0.025669461116194725, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07497569918632507, "sampling/sampling_logp_difference/max": 1.1182861328125, "sampling/importance_sampling_ratio/min": 0.32683947682380676, "sampling/importance_sampling_ratio/mean": 1.0032719373703003, "sampling/importance_sampling_ratio/max": 1.6251786947250366, "entropy": 0.3919231668114662, "clip_ratio/low_mean": 0.015339756850153208, "clip_ratio/low_min": 0.015339756850153208, "clip_ratio/high_mean": 0.07252875994890928, "clip_ratio/high_max": 0.07252875994890928, "clip_ratio/region_mean": 0.08786851679906249, "reward_total_mean": 0.8076785206794739, "reward_meter_mean": 0.989794135093689, "reward_meter_std": 0.00534153962507844, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9452117085456848, "reward_repeat_soft_std": 0.03241921588778496, "reward_judge_quality_mean": 0.39249998331069946, "reward_judge_quality_std": 0.08892211318016052, "reward_total_composite_mean": 0.8076785206794739, "reward_total_composite_std": 0.025669461116194725} {"timestamp_utc": "2026-04-13T05:57:34Z", "mode": "train", "global_step": 3037, "epoch": 0.3050728277247614, "loss": 0.0087, "grad_norm": 7.800347328186035, "learning_rate": 8.000000000000001e-07, "num_tokens": 5685439.0, "completions/mean_length": 127.25, "completions/min_length": 119.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.25, "completions/min_terminated_length": 119.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.9154038429260254, "rewards/meter/std": 0.21296632289886475, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9333134889602661, "rewards/repeat_soft/std": 0.029098689556121826, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7685130834579468, "rewards/total_composite/std": 0.09292757511138916, "reward": 0.7685130834579468, "reward_std": 0.09292755275964737, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10264944285154343, "sampling/sampling_logp_difference/max": 2.91914963722229, "sampling/importance_sampling_ratio/min": 0.05397957190871239, "sampling/importance_sampling_ratio/mean": 0.9929376840591431, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.46385640650987625, "clip_ratio/low_mean": 0.03757416922599077, "clip_ratio/low_min": 0.03757416922599077, "clip_ratio/high_mean": 0.059865465853363276, "clip_ratio/high_max": 0.059865465853363276, "clip_ratio/region_mean": 0.09743963507935405, "reward_total_mean": 0.7685130834579468, "reward_meter_mean": 0.9154038429260254, "reward_meter_std": 0.21296632289886475, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9333134889602661, "reward_repeat_soft_std": 0.029098689556121826, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7685130834579468, "reward_total_composite_std": 0.09292757511138916} {"timestamp_utc": "2026-04-13T05:57:42Z", "mode": "train", "global_step": 3038, "epoch": 0.3051732797589151, "loss": -0.024, "grad_norm": 7.885588645935059, "learning_rate": 7.96969696969697e-07, "num_tokens": 5687756.0, "completions/mean_length": 107.625, "completions/min_length": 87.0, "completions/max_length": 119.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.625, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.9935914278030396, "rewards/meter/std": 0.003222607308998704, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9788512587547302, "rewards/repeat_soft/std": 0.015001500956714153, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.8710012435913086, "rewards/total_composite/std": 0.08451244235038757, "reward": 0.8710012435913086, "reward_std": 0.08451243489980698, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08506835252046585, "sampling/sampling_logp_difference/max": 1.8529090881347656, "sampling/importance_sampling_ratio/min": 0.156780406832695, "sampling/importance_sampling_ratio/mean": 1.0035483837127686, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37883204966783524, "clip_ratio/low_mean": 0.036822851514443755, "clip_ratio/low_min": 0.036822851514443755, "clip_ratio/high_mean": 0.03297746181488037, "clip_ratio/high_max": 0.03297746181488037, "clip_ratio/region_mean": 0.06980031332932413, "reward_total_mean": 0.8710012435913086, "reward_meter_mean": 0.9935914278030396, "reward_meter_std": 0.003222607308998704, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9788512587547302, "reward_repeat_soft_std": 0.015001500956714153, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.8710012435913086, "reward_total_composite_std": 0.08451244235038757} {"timestamp_utc": "2026-04-13T05:57:55Z", "mode": "train", "global_step": 3039, "epoch": 0.3052737317930688, "loss": -0.128, "grad_norm": 0.8643274903297424, "learning_rate": 7.939393939393939e-07, "num_tokens": 5689340.0, "completions/mean_length": 234.0, "completions/min_length": 62.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 67.20000457763672, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.595712423324585, "rewards/meter/std": 0.49339720606803894, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.5175492167472839, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.878768801689148, "rewards/repeat_soft/std": 0.11957699805498123, "rewards/judge_quality/mean": 0.1862500011920929, "rewards/judge_quality/std": 0.163788840174675, "rewards/total_composite/mean": 0.4643225073814392, "rewards/total_composite/std": 0.3858076333999634, "reward": 0.4643225073814392, "reward_std": 0.385807603597641, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.036146510392427444, "sampling/sampling_logp_difference/max": 0.7376446723937988, "sampling/importance_sampling_ratio/min": 0.4782389998435974, "sampling/importance_sampling_ratio/mean": 1.0024206638336182, "sampling/importance_sampling_ratio/max": 1.9489283561706543, "entropy": 0.14516379311680794, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.02235657477285713, "clip_ratio/high_max": 0.02235657477285713, "clip_ratio/region_mean": 0.02235657477285713, "reward_total_mean": 0.4643225073814392, "reward_meter_mean": 0.595712423324585, "reward_meter_std": 0.49339720606803894, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.5175492167472839, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.878768801689148, "reward_repeat_soft_std": 0.11957699805498123, "reward_judge_quality_mean": 0.1862500011920929, "reward_judge_quality_std": 0.163788840174675, "reward_total_composite_mean": 0.4643225073814392, "reward_total_composite_std": 0.3858076333999634} {"timestamp_utc": "2026-04-13T05:58:01Z", "mode": "train", "global_step": 3040, "epoch": 0.3053741838272225, "loss": 0.0027, "grad_norm": 9.239105224609375, "learning_rate": 7.909090909090909e-07, "num_tokens": 5690717.0, "completions/mean_length": 32.125, "completions/min_length": 31.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.125, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9933246374130249, "rewards/meter/std": 0.0022186373826116323, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.46000000834465027, "rewards/judge_quality/std": 0.21138995885849, "rewards/total_composite/mean": 0.8312460780143738, "rewards/total_composite/std": 0.06377655267715454, "reward": 0.8312460780143738, "reward_std": 0.06377656757831573, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1007205918431282, "sampling/sampling_logp_difference/max": 1.1550240516662598, "sampling/importance_sampling_ratio/min": 0.3150499761104584, "sampling/importance_sampling_ratio/mean": 1.005293607711792, "sampling/importance_sampling_ratio/max": 1.8722704648971558, "entropy": 0.6199056506156921, "clip_ratio/low_mean": 0.07332389103248715, "clip_ratio/low_min": 0.07332389103248715, "clip_ratio/high_mean": 0.015625, "clip_ratio/high_max": 0.015625, "clip_ratio/region_mean": 0.08894889103248715, "reward_total_mean": 0.8312460780143738, "reward_meter_mean": 0.9933246374130249, "reward_meter_std": 0.0022186373826116323, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.46000000834465027, "reward_judge_quality_std": 0.21138995885849, "reward_total_composite_mean": 0.8312460780143738, "reward_total_composite_std": 0.06377655267715454} {"timestamp_utc": "2026-04-13T05:58:09Z", "mode": "train", "global_step": 3041, "epoch": 0.3054746358613762, "loss": 0.029, "grad_norm": 12.333770751953125, "learning_rate": 7.878787878787879e-07, "num_tokens": 5692599.0, "completions/mean_length": 55.25, "completions/min_length": 52.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.25, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.8722159266471863, "rewards/meter/std": 0.24859002232551575, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9921917915344238, "rewards/repeat_soft/std": 0.004755111411213875, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.7800912857055664, "rewards/total_composite/std": 0.13174016773700714, "reward": 0.7800912857055664, "reward_std": 0.13174015283584595, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09436895698308945, "sampling/sampling_logp_difference/max": 1.3355209827423096, "sampling/importance_sampling_ratio/min": 0.2630211114883423, "sampling/importance_sampling_ratio/mean": 0.9988486170768738, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4402962289750576, "clip_ratio/low_mean": 0.03767806384712458, "clip_ratio/low_min": 0.03767806384712458, "clip_ratio/high_mean": 0.0643250453285873, "clip_ratio/high_max": 0.0643250453285873, "clip_ratio/region_mean": 0.10200310917571187, "reward_total_mean": 0.7800912857055664, "reward_meter_mean": 0.8722159266471863, "reward_meter_std": 0.24859002232551575, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9921917915344238, "reward_repeat_soft_std": 0.004755111411213875, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.7800912857055664, "reward_total_composite_std": 0.13174016773700714} {"timestamp_utc": "2026-04-13T05:58:21Z", "mode": "train", "global_step": 3042, "epoch": 0.3055750878955299, "loss": -0.1604, "grad_norm": 1.8719078302383423, "learning_rate": 7.84848484848485e-07, "num_tokens": 5694236.0, "completions/mean_length": 121.625, "completions/min_length": 63.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 65.85714721679688, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.8659685850143433, "rewards/meter/std": 0.3254860043525696, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9819655418395996, "rewards/repeat_soft/std": 0.01388065330684185, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.23445606231689453, "rewards/total_composite/mean": 0.7321146726608276, "rewards/total_composite/std": 0.2987108826637268, "reward": 0.7321146726608276, "reward_std": 0.2987108826637268, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10937745124101639, "sampling/sampling_logp_difference/max": 2.880876302719116, "sampling/importance_sampling_ratio/min": 0.05608559027314186, "sampling/importance_sampling_ratio/mean": 1.0018630027770996, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4352485090494156, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08765995409339666, "clip_ratio/high_max": 0.08765995409339666, "clip_ratio/region_mean": 0.08765995409339666, "reward_total_mean": 0.7321146726608276, "reward_meter_mean": 0.8659685850143433, "reward_meter_std": 0.3254860043525696, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9819655418395996, "reward_repeat_soft_std": 0.01388065330684185, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.23445606231689453, "reward_total_composite_mean": 0.7321146726608276, "reward_total_composite_std": 0.2987108826637268} {"timestamp_utc": "2026-04-13T05:58:29Z", "mode": "train", "global_step": 3043, "epoch": 0.30567553992968355, "loss": 0.018, "grad_norm": 7.891928195953369, "learning_rate": 7.818181818181819e-07, "num_tokens": 5696118.0, "completions/mean_length": 66.25, "completions/min_length": 64.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.25, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9892381429672241, "rewards/meter/std": 0.006045436952263117, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9515978097915649, "rewards/repeat_soft/std": 0.027627740055322647, "rewards/judge_quality/mean": 0.5575000047683716, "rewards/judge_quality/std": 0.19955310225486755, "rewards/total_composite/mean": 0.8575669527053833, "rewards/total_composite/std": 0.05759085714817047, "reward": 0.8575669527053833, "reward_std": 0.057590849697589874, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07458963990211487, "sampling/sampling_logp_difference/max": 1.3447856903076172, "sampling/importance_sampling_ratio/min": 0.26059556007385254, "sampling/importance_sampling_ratio/mean": 1.010774850845337, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39826927706599236, "clip_ratio/low_mean": 0.04845172958448529, "clip_ratio/low_min": 0.04845172958448529, "clip_ratio/high_mean": 0.015387347200885415, "clip_ratio/high_max": 0.015387347200885415, "clip_ratio/region_mean": 0.06383907678537071, "reward_total_mean": 0.8575669527053833, "reward_meter_mean": 0.9892381429672241, "reward_meter_std": 0.006045436952263117, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9515978097915649, "reward_repeat_soft_std": 0.027627740055322647, "reward_judge_quality_mean": 0.5575000047683716, "reward_judge_quality_std": 0.19955310225486755, "reward_total_composite_mean": 0.8575669527053833, "reward_total_composite_std": 0.05759085714817047} {"timestamp_utc": "2026-04-13T05:58:38Z", "mode": "train", "global_step": 3044, "epoch": 0.30577599196383726, "loss": -0.0194, "grad_norm": 5.845788478851318, "learning_rate": 7.787878787878788e-07, "num_tokens": 5698607.0, "completions/mean_length": 131.125, "completions/min_length": 113.0, "completions/max_length": 146.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.125, "completions/min_terminated_length": 113.0, "completions/max_terminated_length": 146.0, "rewards/meter/mean": 0.8961887359619141, "rewards/meter/std": 0.0919935554265976, "rewards/count_adherence/mean": 0.8500000238418579, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8433545827865601, "rewards/repeat_soft/std": 0.11977483332157135, "rewards/judge_quality/mean": 0.5575000047683716, "rewards/judge_quality/std": 0.19955310225486755, "rewards/total_composite/mean": 0.7823703289031982, "rewards/total_composite/std": 0.05330402031540871, "reward": 0.7823703289031982, "reward_std": 0.053303997963666916, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08324578404426575, "sampling/sampling_logp_difference/max": 2.9272093772888184, "sampling/importance_sampling_ratio/min": 0.05354625731706619, "sampling/importance_sampling_ratio/mean": 0.9922420978546143, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.24471492692828178, "clip_ratio/low_mean": 0.02689205203205347, "clip_ratio/low_min": 0.02689205203205347, "clip_ratio/high_mean": 0.037121754605323076, "clip_ratio/high_max": 0.037121754605323076, "clip_ratio/region_mean": 0.06401380663737655, "reward_total_mean": 0.7823703289031982, "reward_meter_mean": 0.8961887359619141, "reward_meter_std": 0.0919935554265976, "reward_count_adherence_mean": 0.8500000238418579, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8433545827865601, "reward_repeat_soft_std": 0.11977483332157135, "reward_judge_quality_mean": 0.5575000047683716, "reward_judge_quality_std": 0.19955310225486755, "reward_total_composite_mean": 0.7823703289031982, "reward_total_composite_std": 0.05330402031540871} {"timestamp_utc": "2026-04-13T05:58:44Z", "mode": "train", "global_step": 3045, "epoch": 0.30587644399799097, "loss": 0.0, "grad_norm": 8.528696060180664, "learning_rate": 7.757575757575758e-07, "num_tokens": 5700379.0, "completions/mean_length": 64.5, "completions/min_length": 60.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.5, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9939543008804321, "rewards/meter/std": 0.000998802948743105, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9552737474441528, "rewards/repeat_soft/std": 0.02594299428164959, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8188068270683289, "rewards/total_composite/std": 0.002602348569780588, "reward": 0.8188068270683289, "reward_std": 0.002602341352030635, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09730955213308334, "sampling/sampling_logp_difference/max": 2.55433988571167, "sampling/importance_sampling_ratio/min": 0.0777435302734375, "sampling/importance_sampling_ratio/mean": 0.9908545613288879, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3663884475827217, "clip_ratio/low_mean": 0.013906740467064083, "clip_ratio/low_min": 0.013906740467064083, "clip_ratio/high_mean": 0.045683009549975395, "clip_ratio/high_max": 0.045683009549975395, "clip_ratio/region_mean": 0.05958975001703948, "reward_total_mean": 0.8188068270683289, "reward_meter_mean": 0.9939543008804321, "reward_meter_std": 0.000998802948743105, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9552737474441528, "reward_repeat_soft_std": 0.02594299428164959, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8188068270683289, "reward_total_composite_std": 0.002602348569780588} {"timestamp_utc": "2026-04-13T05:58:52Z", "mode": "train", "global_step": 3046, "epoch": 0.3059768960321447, "loss": 0.0201, "grad_norm": 5.2410383224487305, "learning_rate": 7.727272727272727e-07, "num_tokens": 5703092.0, "completions/mean_length": 149.125, "completions/min_length": 144.0, "completions/max_length": 158.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 149.125, "completions/min_terminated_length": 144.0, "completions/max_terminated_length": 158.0, "rewards/meter/mean": 0.9910992383956909, "rewards/meter/std": 0.0009512970573268831, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8817901611328125, "rewards/repeat_soft/std": 0.040374089032411575, "rewards/judge_quality/mean": 0.3137499690055847, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7782986760139465, "rewards/total_composite/std": 0.025523196905851364, "reward": 0.7782986760139465, "reward_std": 0.025523191317915916, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07229612022638321, "sampling/sampling_logp_difference/max": 2.463439702987671, "sampling/importance_sampling_ratio/min": 0.08514158427715302, "sampling/importance_sampling_ratio/mean": 1.0078656673431396, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36014988645911217, "clip_ratio/low_mean": 0.04719669558107853, "clip_ratio/low_min": 0.04719669558107853, "clip_ratio/high_mean": 0.030932333320379257, "clip_ratio/high_max": 0.030932333320379257, "clip_ratio/region_mean": 0.07812902890145779, "reward_total_mean": 0.7782986760139465, "reward_meter_mean": 0.9910992383956909, "reward_meter_std": 0.0009512970573268831, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8817901611328125, "reward_repeat_soft_std": 0.040374089032411575, "reward_judge_quality_mean": 0.3137499690055847, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7782986760139465, "reward_total_composite_std": 0.025523196905851364} {"timestamp_utc": "2026-04-13T05:59:00Z", "mode": "train", "global_step": 3047, "epoch": 0.3060773480662983, "loss": 0.0108, "grad_norm": 6.04510498046875, "learning_rate": 7.696969696969698e-07, "num_tokens": 5705360.0, "completions/mean_length": 112.5, "completions/min_length": 105.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.5, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.9756618738174438, "rewards/meter/std": 0.05392264202237129, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8788742423057556, "rewards/repeat_soft/std": 0.026628883555531502, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.836685299873352, "rewards/total_composite/std": 0.06023499369621277, "reward": 0.836685299873352, "reward_std": 0.06023499369621277, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05363292992115021, "sampling/sampling_logp_difference/max": 1.5716652870178223, "sampling/importance_sampling_ratio/min": 0.2076990157365799, "sampling/importance_sampling_ratio/mean": 1.0043772459030151, "sampling/importance_sampling_ratio/max": 1.9656982421875, "entropy": 0.27143662609159946, "clip_ratio/low_mean": 0.034875806886702776, "clip_ratio/low_min": 0.034875806886702776, "clip_ratio/high_mean": 0.01801948039792478, "clip_ratio/high_max": 0.01801948039792478, "clip_ratio/region_mean": 0.05289528728462756, "reward_total_mean": 0.836685299873352, "reward_meter_mean": 0.9756618738174438, "reward_meter_std": 0.05392264202237129, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8788742423057556, "reward_repeat_soft_std": 0.026628883555531502, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.836685299873352, "reward_total_composite_std": 0.06023499369621277} {"timestamp_utc": "2026-04-13T05:59:07Z", "mode": "train", "global_step": 3048, "epoch": 0.30617780010045204, "loss": 0.0309, "grad_norm": 9.643020629882812, "learning_rate": 7.666666666666667e-07, "num_tokens": 5707208.0, "completions/mean_length": 58.0, "completions/min_length": 55.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.985123872756958, "rewards/meter/std": 0.004340935032814741, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9833382368087769, "rewards/repeat_soft/std": 0.009416146203875542, "rewards/judge_quality/mean": 0.5950000286102295, "rewards/judge_quality/std": 0.19820626080036163, "rewards/total_composite/mean": 0.8701395988464355, "rewards/total_composite/std": 0.059336405247449875, "reward": 0.8701395988464355, "reward_std": 0.05933639407157898, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06374507397413254, "sampling/sampling_logp_difference/max": 1.1362299919128418, "sampling/importance_sampling_ratio/min": 0.321027010679245, "sampling/importance_sampling_ratio/mean": 1.0204740762710571, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.28724910877645016, "clip_ratio/low_mean": 0.03782325237989426, "clip_ratio/low_min": 0.03782325237989426, "clip_ratio/high_mean": 0.01753246784210205, "clip_ratio/high_max": 0.01753246784210205, "clip_ratio/region_mean": 0.05535572022199631, "reward_total_mean": 0.8701395988464355, "reward_meter_mean": 0.985123872756958, "reward_meter_std": 0.004340935032814741, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9833382368087769, "reward_repeat_soft_std": 0.009416146203875542, "reward_judge_quality_mean": 0.5950000286102295, "reward_judge_quality_std": 0.19820626080036163, "reward_total_composite_mean": 0.8701395988464355, "reward_total_composite_std": 0.059336405247449875} {"timestamp_utc": "2026-04-13T05:59:13Z", "mode": "train", "global_step": 3049, "epoch": 0.30627825213460574, "loss": -0.002, "grad_norm": 9.896900177001953, "learning_rate": 7.636363636363637e-07, "num_tokens": 5709103.0, "completions/mean_length": 44.875, "completions/min_length": 44.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.875, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9474133253097534, "rewards/meter/std": 0.0323379710316658, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8730234503746033, "rewards/repeat_soft/std": 0.1110394224524498, "rewards/judge_quality/mean": 0.32499998807907104, "rewards/judge_quality/std": 0.13887301087379456, "rewards/total_composite/mean": 0.7611383199691772, "rewards/total_composite/std": 0.04115898162126541, "reward": 0.7611383199691772, "reward_std": 0.04115897789597511, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.056854888796806335, "sampling/sampling_logp_difference/max": 1.3150436878204346, "sampling/importance_sampling_ratio/min": 0.31145405769348145, "sampling/importance_sampling_ratio/mean": 0.9951501488685608, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.22783544100821018, "clip_ratio/low_mean": 0.0027777778450399637, "clip_ratio/low_min": 0.0027777778450399637, "clip_ratio/high_mean": 0.027209596941247582, "clip_ratio/high_max": 0.027209596941247582, "clip_ratio/region_mean": 0.029987374786287546, "reward_total_mean": 0.7611383199691772, "reward_meter_mean": 0.9474133253097534, "reward_meter_std": 0.0323379710316658, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8730234503746033, "reward_repeat_soft_std": 0.1110394224524498, "reward_judge_quality_mean": 0.32499998807907104, "reward_judge_quality_std": 0.13887301087379456, "reward_total_composite_mean": 0.7611383199691772, "reward_total_composite_std": 0.04115898162126541} {"timestamp_utc": "2026-04-13T05:59:20Z", "mode": "train", "global_step": 3050, "epoch": 0.3063787041687594, "loss": 0.0783, "grad_norm": 8.236157417297363, "learning_rate": 7.606060606060607e-07, "num_tokens": 5711582.0, "completions/mean_length": 132.875, "completions/min_length": 116.0, "completions/max_length": 163.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.875, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 163.0, "rewards/meter/mean": 0.7583379745483398, "rewards/meter/std": 0.17417925596237183, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.08625820279121399, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.838050901889801, "rewards/repeat_soft/std": 0.06940924376249313, "rewards/judge_quality/mean": 0.5399999618530273, "rewards/judge_quality/std": 0.22245386242866516, "rewards/total_composite/mean": 0.7276821732521057, "rewards/total_composite/std": 0.046055302023887634, "reward": 0.7276821732521057, "reward_std": 0.04605530574917793, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10064057260751724, "sampling/sampling_logp_difference/max": 1.8252637386322021, "sampling/importance_sampling_ratio/min": 0.16117513179779053, "sampling/importance_sampling_ratio/mean": 1.0018339157104492, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3548012003302574, "clip_ratio/low_mean": 0.0510159395635128, "clip_ratio/low_min": 0.0510159395635128, "clip_ratio/high_mean": 0.034281056839972734, "clip_ratio/high_max": 0.034281056839972734, "clip_ratio/region_mean": 0.08529699640348554, "reward_total_mean": 0.7276821732521057, "reward_meter_mean": 0.7583379745483398, "reward_meter_std": 0.17417925596237183, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.08625820279121399, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.838050901889801, "reward_repeat_soft_std": 0.06940924376249313, "reward_judge_quality_mean": 0.5399999618530273, "reward_judge_quality_std": 0.22245386242866516, "reward_total_composite_mean": 0.7276821732521057, "reward_total_composite_std": 0.046055302023887634} {"timestamp_utc": "2026-04-13T06:00:14Z", "mode": "eval", "global_step": 3050, "epoch": 0.3063787041687594, "eval_loss": NaN, "eval_runtime": 52.9933, "eval_samples_per_second": 1.51, "eval_steps_per_second": 0.189, "eval_num_tokens": 5711582.0, "eval_completions/mean_length": 104.2625, "eval_completions/min_length": 42.8, "eval_completions/max_length": 234.3, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 94.08571548461914, "eval_completions/min_terminated_length": 42.8, "eval_completions/max_terminated_length": 163.0, "eval_rewards/meter/mean": 0.9137624323368072, "eval_rewards/meter/std": 0.15195817914791404, "eval_rewards/count_adherence/mean": 0.9897916674613952, "eval_rewards/count_adherence/std": 0.02887352667748928, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.07071067690849304, "eval_rewards/repeat_soft/mean": 0.9090190887451172, "eval_rewards/repeat_soft/std": 0.07921404242515565, "eval_rewards/judge_quality/mean": 0.42950000762939455, "eval_rewards/judge_quality/std": 0.1441200479865074, "eval_rewards/total_composite/mean": 0.7667000234127045, "eval_rewards/total_composite/std": 0.12923006135970355, "eval_reward": 0.7667000234127045, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.036175782606005666, "eval_sampling/sampling_logp_difference/max": 0.8201593399047852, "eval_sampling/importance_sampling_ratio/min": 0.4521248430013657, "eval_sampling/importance_sampling_ratio/mean": 1.0068567037582397, "eval_sampling/importance_sampling_ratio/max": 1.3062537074089051, "eval_entropy": 0.3647799462080002, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7667000234127045, "eval_reward_meter_mean": 0.9137624323368072, "eval_reward_meter_std": 0.15195817914791404, "eval_reward_count_adherence_mean": 0.9897916674613952, "eval_reward_count_adherence_std": 0.02887352667748928, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.07071067690849304, "eval_reward_repeat_soft_mean": 0.9090190887451172, "eval_reward_repeat_soft_std": 0.07921404242515565, "eval_reward_judge_quality_mean": 0.42950000762939455, "eval_reward_judge_quality_std": 0.1441200479865074, "eval_reward_total_composite_mean": 0.7667000234127045, "eval_reward_total_composite_std": 0.12923006135970355} {"timestamp_utc": "2026-04-13T06:00:24Z", "mode": "train", "global_step": 3051, "epoch": 0.3064791562029131, "loss": 0.0086, "grad_norm": 6.354383945465088, "learning_rate": 7.575757575757576e-07, "num_tokens": 5713726.0, "completions/mean_length": 106.0, "completions/min_length": 99.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.0, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.9930424094200134, "rewards/meter/std": 0.0026636284310370684, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8655825257301331, "rewards/repeat_soft/std": 0.04401512071490288, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7966773509979248, "rewards/total_composite/std": 0.026023315265774727, "reward": 0.7966773509979248, "reward_std": 0.02602331154048443, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0760253518819809, "sampling/sampling_logp_difference/max": 1.994767189025879, "sampling/importance_sampling_ratio/min": 0.13604532182216644, "sampling/importance_sampling_ratio/mean": 1.001550316810608, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.34235259890556335, "clip_ratio/low_mean": 0.012100168038159609, "clip_ratio/low_min": 0.012100168038159609, "clip_ratio/high_mean": 0.06550715351477265, "clip_ratio/high_max": 0.06550715351477265, "clip_ratio/region_mean": 0.07760732155293226, "reward_total_mean": 0.7966773509979248, "reward_meter_mean": 0.9930424094200134, "reward_meter_std": 0.0026636284310370684, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8655825257301331, "reward_repeat_soft_std": 0.04401512071490288, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7966773509979248, "reward_total_composite_std": 0.026023315265774727} {"timestamp_utc": "2026-04-13T06:00:31Z", "mode": "train", "global_step": 3052, "epoch": 0.3065796082370668, "loss": 0.0116, "grad_norm": 7.607499122619629, "learning_rate": 7.545454545454546e-07, "num_tokens": 5716022.0, "completions/mean_length": 107.0, "completions/min_length": 99.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.0, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.9884234666824341, "rewards/meter/std": 0.008195240050554276, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9106128215789795, "rewards/repeat_soft/std": 0.045002344995737076, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8118518590927124, "rewards/total_composite/std": 0.00505440728738904, "reward": 0.8118518590927124, "reward_std": 0.0050544096156954765, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10013706237077713, "sampling/sampling_logp_difference/max": 1.3479368686676025, "sampling/importance_sampling_ratio/min": 0.25977566838264465, "sampling/importance_sampling_ratio/mean": 0.9979504346847534, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4678415581583977, "clip_ratio/low_mean": 0.025859427638351917, "clip_ratio/low_min": 0.025859427638351917, "clip_ratio/high_mean": 0.05835666228085756, "clip_ratio/high_max": 0.05835666228085756, "clip_ratio/region_mean": 0.08421608991920948, "reward_total_mean": 0.8118518590927124, "reward_meter_mean": 0.9884234666824341, "reward_meter_std": 0.008195240050554276, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9106128215789795, "reward_repeat_soft_std": 0.045002344995737076, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8118518590927124, "reward_total_composite_std": 0.00505440728738904} {"timestamp_utc": "2026-04-13T06:00:38Z", "mode": "train", "global_step": 3053, "epoch": 0.30668006027122047, "loss": 0.0092, "grad_norm": 10.640708923339844, "learning_rate": 7.515151515151516e-07, "num_tokens": 5717408.0, "completions/mean_length": 23.25, "completions/min_length": 20.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.25, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.985321581363678, "rewards/meter/std": 0.0036246473900973797, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9612975120544434, "rewards/repeat_soft/std": 0.0034009767696261406, "rewards/judge_quality/mean": 0.6850000023841858, "rewards/judge_quality/std": 0.2512255907058716, "rewards/total_composite/mean": 0.8950244784355164, "rewards/total_composite/std": 0.07597630470991135, "reward": 0.8950244784355164, "reward_std": 0.07597629725933075, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07416626811027527, "sampling/sampling_logp_difference/max": 1.1636314392089844, "sampling/importance_sampling_ratio/min": 0.3123498260974884, "sampling/importance_sampling_ratio/mean": 1.00872004032135, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.35573162138462067, "clip_ratio/low_mean": 0.07279680855572224, "clip_ratio/low_min": 0.07279680855572224, "clip_ratio/high_mean": 0.015434782486408949, "clip_ratio/high_max": 0.015434782486408949, "clip_ratio/region_mean": 0.08823159104213119, "reward_total_mean": 0.8950244784355164, "reward_meter_mean": 0.985321581363678, "reward_meter_std": 0.0036246473900973797, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9612975120544434, "reward_repeat_soft_std": 0.0034009767696261406, "reward_judge_quality_mean": 0.6850000023841858, "reward_judge_quality_std": 0.2512255907058716, "reward_total_composite_mean": 0.8950244784355164, "reward_total_composite_std": 0.07597630470991135} {"timestamp_utc": "2026-04-13T06:00:45Z", "mode": "train", "global_step": 3054, "epoch": 0.3067805123053742, "loss": 0.0377, "grad_norm": 15.883240699768066, "learning_rate": 7.484848484848485e-07, "num_tokens": 5718845.0, "completions/mean_length": 30.625, "completions/min_length": 29.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.625, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.7158949375152588, "rewards/meter/std": 0.38385242223739624, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9274048805236816, "rewards/repeat_soft/std": 0.06356074661016464, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.6931432485580444, "rewards/total_composite/std": 0.17774611711502075, "reward": 0.6931432485580444, "reward_std": 0.17774611711502075, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09851796180009842, "sampling/sampling_logp_difference/max": 1.1110620498657227, "sampling/importance_sampling_ratio/min": 0.3292091488838196, "sampling/importance_sampling_ratio/mean": 1.018250823020935, "sampling/importance_sampling_ratio/max": 1.7827720642089844, "entropy": 0.510151170194149, "clip_ratio/low_mean": 0.040616599842906, "clip_ratio/low_min": 0.040616599842906, "clip_ratio/high_mean": 0.061772630084306, "clip_ratio/high_max": 0.061772630084306, "clip_ratio/region_mean": 0.102389229927212, "reward_total_mean": 0.6931432485580444, "reward_meter_mean": 0.7158949375152588, "reward_meter_std": 0.38385242223739624, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9274048805236816, "reward_repeat_soft_std": 0.06356074661016464, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.6931432485580444, "reward_total_composite_std": 0.17774611711502075} {"timestamp_utc": "2026-04-13T06:00:58Z", "mode": "train", "global_step": 3055, "epoch": 0.3068809643395279, "loss": 0.0537, "grad_norm": 10.186320304870605, "learning_rate": 7.454545454545455e-07, "num_tokens": 5720539.0, "completions/mean_length": 57.75, "completions/min_length": 55.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.75, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9000576138496399, "rewards/meter/std": 0.14952056109905243, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9727513790130615, "rewards/repeat_soft/std": 0.026926716789603233, "rewards/judge_quality/mean": 0.5049999952316284, "rewards/judge_quality/std": 0.16801361739635468, "rewards/total_composite/mean": 0.8038010597229004, "rewards/total_composite/std": 0.09360961616039276, "reward": 0.8038010597229004, "reward_std": 0.09360961616039276, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0879639983177185, "sampling/sampling_logp_difference/max": 1.4062716960906982, "sampling/importance_sampling_ratio/min": 0.2450552135705948, "sampling/importance_sampling_ratio/mean": 1.0064504146575928, "sampling/importance_sampling_ratio/max": 1.7812278270721436, "entropy": 0.47251151874661446, "clip_ratio/low_mean": 0.035527005791664124, "clip_ratio/low_min": 0.035527005791664124, "clip_ratio/high_mean": 0.05527047486975789, "clip_ratio/high_max": 0.05527047486975789, "clip_ratio/region_mean": 0.09079748066142201, "reward_total_mean": 0.8038010597229004, "reward_meter_mean": 0.9000576138496399, "reward_meter_std": 0.14952056109905243, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9727513790130615, "reward_repeat_soft_std": 0.026926716789603233, "reward_judge_quality_mean": 0.5049999952316284, "reward_judge_quality_std": 0.16801361739635468, "reward_total_composite_mean": 0.8038010597229004, "reward_total_composite_std": 0.09360961616039276} {"timestamp_utc": "2026-04-13T06:01:06Z", "mode": "train", "global_step": 3056, "epoch": 0.3069814163736816, "loss": 0.0094, "grad_norm": 6.180540084838867, "learning_rate": 7.424242424242425e-07, "num_tokens": 5723317.0, "completions/mean_length": 159.25, "completions/min_length": 149.0, "completions/max_length": 177.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 159.25, "completions/min_terminated_length": 149.0, "completions/max_terminated_length": 177.0, "rewards/meter/mean": 0.9641280174255371, "rewards/meter/std": 0.06230047717690468, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.936205267906189, "rewards/repeat_soft/std": 0.045796170830726624, "rewards/judge_quality/mean": 0.25874999165534973, "rewards/judge_quality/std": 0.07395702600479126, "rewards/total_composite/mean": 0.7551031112670898, "rewards/total_composite/std": 0.0387696772813797, "reward": 0.7551031112670898, "reward_std": 0.0387696735560894, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09425266087055206, "sampling/sampling_logp_difference/max": 1.5995006561279297, "sampling/importance_sampling_ratio/min": 0.20199736952781677, "sampling/importance_sampling_ratio/mean": 1.0094612836837769, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5001130625605583, "clip_ratio/low_mean": 0.027761003002524376, "clip_ratio/low_min": 0.027761003002524376, "clip_ratio/high_mean": 0.051060983911156654, "clip_ratio/high_max": 0.051060983911156654, "clip_ratio/region_mean": 0.07882198691368103, "reward_total_mean": 0.7551031112670898, "reward_meter_mean": 0.9641280174255371, "reward_meter_std": 0.06230047717690468, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.936205267906189, "reward_repeat_soft_std": 0.045796170830726624, "reward_judge_quality_mean": 0.25874999165534973, "reward_judge_quality_std": 0.07395702600479126, "reward_total_composite_mean": 0.7551031112670898, "reward_total_composite_std": 0.0387696772813797} {"timestamp_utc": "2026-04-13T06:01:12Z", "mode": "train", "global_step": 3057, "epoch": 0.30708186840783525, "loss": 0.0086, "grad_norm": 15.5534029006958, "learning_rate": 7.393939393939395e-07, "num_tokens": 5724698.0, "completions/mean_length": 28.625, "completions/min_length": 25.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.625, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.8511934280395508, "rewards/meter/std": 0.26429909467697144, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9065572023391724, "rewards/repeat_soft/std": 0.038421694189310074, "rewards/judge_quality/mean": 0.39375001192092896, "rewards/judge_quality/std": 0.09941794723272324, "rewards/total_composite/mean": 0.741817831993103, "rewards/total_composite/std": 0.11493292450904846, "reward": 0.741817831993103, "reward_std": 0.11493291705846786, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08420581370592117, "sampling/sampling_logp_difference/max": 1.4776203632354736, "sampling/importance_sampling_ratio/min": 0.22818003594875336, "sampling/importance_sampling_ratio/mean": 0.9938065409660339, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40962161868810654, "clip_ratio/low_mean": 0.012931034434586763, "clip_ratio/low_min": 0.012931034434586763, "clip_ratio/high_mean": 0.07869828818365932, "clip_ratio/high_max": 0.07869828818365932, "clip_ratio/region_mean": 0.09162932261824608, "reward_total_mean": 0.741817831993103, "reward_meter_mean": 0.8511934280395508, "reward_meter_std": 0.26429909467697144, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9065572023391724, "reward_repeat_soft_std": 0.038421694189310074, "reward_judge_quality_mean": 0.39375001192092896, "reward_judge_quality_std": 0.09941794723272324, "reward_total_composite_mean": 0.741817831993103, "reward_total_composite_std": 0.11493292450904846} {"timestamp_utc": "2026-04-13T06:01:19Z", "mode": "train", "global_step": 3058, "epoch": 0.30718232044198895, "loss": -0.0006, "grad_norm": 5.630176544189453, "learning_rate": 7.363636363636364e-07, "num_tokens": 5726820.0, "completions/mean_length": 112.25, "completions/min_length": 99.0, "completions/max_length": 119.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.25, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.97539222240448, "rewards/meter/std": 0.01616223342716694, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8806695938110352, "rewards/repeat_soft/std": 0.045408591628074646, "rewards/judge_quality/mean": 0.27125000953674316, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.753680944442749, "rewards/total_composite/std": 0.026046017184853554, "reward": 0.753680944442749, "reward_std": 0.026046007871627808, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07811237871646881, "sampling/sampling_logp_difference/max": 1.7560460567474365, "sampling/importance_sampling_ratio/min": 0.17272648215293884, "sampling/importance_sampling_ratio/mean": 1.0003660917282104, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31482504308223724, "clip_ratio/low_mean": 0.0288950412068516, "clip_ratio/low_min": 0.0288950412068516, "clip_ratio/high_mean": 0.03415302745997906, "clip_ratio/high_max": 0.03415302745997906, "clip_ratio/region_mean": 0.06304806866683066, "reward_total_mean": 0.753680944442749, "reward_meter_mean": 0.97539222240448, "reward_meter_std": 0.01616223342716694, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8806695938110352, "reward_repeat_soft_std": 0.045408591628074646, "reward_judge_quality_mean": 0.27125000953674316, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.753680944442749, "reward_total_composite_std": 0.026046017184853554} {"timestamp_utc": "2026-04-13T06:01:31Z", "mode": "train", "global_step": 3059, "epoch": 0.30728277247614266, "loss": -0.1937, "grad_norm": 1.3662374019622803, "learning_rate": 7.333333333333334e-07, "num_tokens": 5728950.0, "completions/mean_length": 145.25, "completions/min_length": 91.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 92.85714721679688, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.9636386632919312, "rewards/meter/std": 0.02447454072535038, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.825433611869812, "rewards/repeat_soft/std": 0.0958649069070816, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.6908923983573914, "rewards/total_composite/std": 0.2793271243572235, "reward": 0.6908923983573914, "reward_std": 0.2793271243572235, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06208476051688194, "sampling/sampling_logp_difference/max": 1.6023406982421875, "sampling/importance_sampling_ratio/min": 0.2014244943857193, "sampling/importance_sampling_ratio/mean": 0.9991458654403687, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.271013468503952, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06299386289902031, "clip_ratio/high_max": 0.06299386289902031, "clip_ratio/region_mean": 0.06299386289902031, "reward_total_mean": 0.6908923983573914, "reward_meter_mean": 0.9636386632919312, "reward_meter_std": 0.02447454072535038, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.825433611869812, "reward_repeat_soft_std": 0.0958649069070816, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.6908923983573914, "reward_total_composite_std": 0.2793271243572235} {"timestamp_utc": "2026-04-13T06:01:37Z", "mode": "train", "global_step": 3060, "epoch": 0.3073832245102963, "loss": -0.0195, "grad_norm": 9.569920539855957, "learning_rate": 7.303030303030304e-07, "num_tokens": 5730639.0, "completions/mean_length": 45.125, "completions/min_length": 40.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.125, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9824808239936829, "rewards/meter/std": 0.023205967620015144, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9070438146591187, "rewards/repeat_soft/std": 0.07396112382411957, "rewards/judge_quality/mean": 0.9200000166893005, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.9588208198547363, "rewards/total_composite/std": 0.011739562265574932, "reward": 0.9588208198547363, "reward_std": 0.011739571578800678, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06458359211683273, "sampling/sampling_logp_difference/max": 1.2091565132141113, "sampling/importance_sampling_ratio/min": 0.29844892024993896, "sampling/importance_sampling_ratio/mean": 1.0047695636749268, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31842419877648354, "clip_ratio/low_mean": 0.01793478289619088, "clip_ratio/low_min": 0.01793478289619088, "clip_ratio/high_mean": 0.034084178041666746, "clip_ratio/high_max": 0.034084178041666746, "clip_ratio/region_mean": 0.05201896093785763, "reward_total_mean": 0.9588208198547363, "reward_meter_mean": 0.9824808239936829, "reward_meter_std": 0.023205967620015144, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9070438146591187, "reward_repeat_soft_std": 0.07396112382411957, "reward_judge_quality_mean": 0.9200000166893005, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.9588208198547363, "reward_total_composite_std": 0.011739562265574932} {"timestamp_utc": "2026-04-13T06:01:43Z", "mode": "train", "global_step": 3061, "epoch": 0.30748367654445, "loss": 0.0369, "grad_norm": 9.135332107543945, "learning_rate": 7.272727272727273e-07, "num_tokens": 5732340.0, "completions/mean_length": 44.625, "completions/min_length": 40.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.625, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9205079078674316, "rewards/meter/std": 0.07856788486242294, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9372313022613525, "rewards/repeat_soft/std": 0.08467227965593338, "rewards/judge_quality/mean": 0.476250022649765, "rewards/judge_quality/std": 0.11160357296466827, "rewards/total_composite/mean": 0.8008266687393188, "rewards/total_composite/std": 0.05661793425679207, "reward": 0.8008266687393188, "reward_std": 0.056617945432662964, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09281594306230545, "sampling/sampling_logp_difference/max": 2.325347900390625, "sampling/importance_sampling_ratio/min": 0.09774942696094513, "sampling/importance_sampling_ratio/mean": 1.0196151733398438, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45585213229060173, "clip_ratio/low_mean": 0.04767287336289883, "clip_ratio/low_min": 0.04767287336289883, "clip_ratio/high_mean": 0.05260759778320789, "clip_ratio/high_max": 0.05260759778320789, "clip_ratio/region_mean": 0.10028047114610672, "reward_total_mean": 0.8008266687393188, "reward_meter_mean": 0.9205079078674316, "reward_meter_std": 0.07856788486242294, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9372313022613525, "reward_repeat_soft_std": 0.08467227965593338, "reward_judge_quality_mean": 0.476250022649765, "reward_judge_quality_std": 0.11160357296466827, "reward_total_composite_mean": 0.8008266687393188, "reward_total_composite_std": 0.05661793425679207} {"timestamp_utc": "2026-04-13T06:01:50Z", "mode": "train", "global_step": 3062, "epoch": 0.30758412857860373, "loss": 0.0234, "grad_norm": 9.504039764404297, "learning_rate": 7.242424242424243e-07, "num_tokens": 5734074.0, "completions/mean_length": 54.75, "completions/min_length": 50.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.75, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.988139271736145, "rewards/meter/std": 0.003629326820373535, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9661227464675903, "rewards/repeat_soft/std": 0.04262644797563553, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.821774959564209, "rewards/total_composite/std": 0.008112826384603977, "reward": 0.821774959564209, "reward_std": 0.008112824521958828, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08237883448600769, "sampling/sampling_logp_difference/max": 1.762319564819336, "sampling/importance_sampling_ratio/min": 0.17164625227451324, "sampling/importance_sampling_ratio/mean": 1.018349528312683, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4154408238828182, "clip_ratio/low_mean": 0.04029395803809166, "clip_ratio/low_min": 0.04029395803809166, "clip_ratio/high_mean": 0.027472528163343668, "clip_ratio/high_max": 0.027472528163343668, "clip_ratio/region_mean": 0.06776648620143533, "reward_total_mean": 0.821774959564209, "reward_meter_mean": 0.988139271736145, "reward_meter_std": 0.003629326820373535, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9661227464675903, "reward_repeat_soft_std": 0.04262644797563553, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.821774959564209, "reward_total_composite_std": 0.008112826384603977} {"timestamp_utc": "2026-04-13T06:01:57Z", "mode": "train", "global_step": 3063, "epoch": 0.3076845806127574, "loss": 0.0073, "grad_norm": 5.420004844665527, "learning_rate": 7.212121212121213e-07, "num_tokens": 5736367.0, "completions/mean_length": 124.625, "completions/min_length": 106.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 124.625, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.776698112487793, "rewards/meter/std": 0.31768208742141724, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7592546343803406, "rewards/repeat_soft/std": 0.09884700924158096, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.69675213098526, "rewards/total_composite/std": 0.145114466547966, "reward": 0.69675213098526, "reward_std": 0.1451144814491272, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07248689234256744, "sampling/sampling_logp_difference/max": 1.6560977697372437, "sampling/importance_sampling_ratio/min": 0.19088239967823029, "sampling/importance_sampling_ratio/mean": 0.9977349042892456, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3486923035234213, "clip_ratio/low_mean": 0.024383331183344126, "clip_ratio/low_min": 0.024383331183344126, "clip_ratio/high_mean": 0.05039225239306688, "clip_ratio/high_max": 0.05039225239306688, "clip_ratio/region_mean": 0.07477558357641101, "reward_total_mean": 0.69675213098526, "reward_meter_mean": 0.776698112487793, "reward_meter_std": 0.31768208742141724, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7592546343803406, "reward_repeat_soft_std": 0.09884700924158096, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.69675213098526, "reward_total_composite_std": 0.145114466547966} {"timestamp_utc": "2026-04-13T06:02:05Z", "mode": "train", "global_step": 3064, "epoch": 0.3077850326469111, "loss": 0.0207, "grad_norm": 4.431118488311768, "learning_rate": 7.181818181818182e-07, "num_tokens": 5739405.0, "completions/mean_length": 179.75, "completions/min_length": 175.0, "completions/max_length": 186.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 179.75, "completions/min_terminated_length": 175.0, "completions/max_terminated_length": 186.0, "rewards/meter/mean": 0.991387128829956, "rewards/meter/std": 0.0040714191272854805, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.901401937007904, "rewards/repeat_soft/std": 0.05001634731888771, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.7830144166946411, "rewards/total_composite/std": 0.03304366394877434, "reward": 0.7830144166946411, "reward_std": 0.033043645322322845, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07048378884792328, "sampling/sampling_logp_difference/max": 1.6693754196166992, "sampling/importance_sampling_ratio/min": 0.1883646696805954, "sampling/importance_sampling_ratio/mean": 1.0054950714111328, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3317547421902418, "clip_ratio/low_mean": 0.03547881683334708, "clip_ratio/low_min": 0.03547881683334708, "clip_ratio/high_mean": 0.039067529141902924, "clip_ratio/high_max": 0.039067529141902924, "clip_ratio/region_mean": 0.07454634597525, "reward_total_mean": 0.7830144166946411, "reward_meter_mean": 0.991387128829956, "reward_meter_std": 0.0040714191272854805, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.901401937007904, "reward_repeat_soft_std": 0.05001634731888771, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.7830144166946411, "reward_total_composite_std": 0.03304366394877434} {"timestamp_utc": "2026-04-13T06:02:12Z", "mode": "train", "global_step": 3065, "epoch": 0.3078854846810648, "loss": 0.0272, "grad_norm": 8.584653854370117, "learning_rate": 7.151515151515153e-07, "num_tokens": 5741807.0, "completions/mean_length": 112.25, "completions/min_length": 109.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.25, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.9655986428260803, "rewards/meter/std": 0.041848164051771164, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9550233483314514, "rewards/repeat_soft/std": 0.017061367630958557, "rewards/judge_quality/mean": 0.35999998450279236, "rewards/judge_quality/std": 0.09165150672197342, "rewards/total_composite/mean": 0.7880216836929321, "rewards/total_composite/std": 0.027158543467521667, "reward": 0.7880216836929321, "reward_std": 0.027158521115779877, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10113434493541718, "sampling/sampling_logp_difference/max": 1.9824252128601074, "sampling/importance_sampling_ratio/min": 0.1377348005771637, "sampling/importance_sampling_ratio/mean": 0.9875171184539795, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4725642018020153, "clip_ratio/low_mean": 0.03126932727172971, "clip_ratio/low_min": 0.03126932727172971, "clip_ratio/high_mean": 0.058904074132442474, "clip_ratio/high_max": 0.058904074132442474, "clip_ratio/region_mean": 0.09017340140417218, "reward_total_mean": 0.7880216836929321, "reward_meter_mean": 0.9655986428260803, "reward_meter_std": 0.041848164051771164, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9550233483314514, "reward_repeat_soft_std": 0.017061367630958557, "reward_judge_quality_mean": 0.35999998450279236, "reward_judge_quality_std": 0.09165150672197342, "reward_total_composite_mean": 0.7880216836929321, "reward_total_composite_std": 0.027158543467521667} {"timestamp_utc": "2026-04-13T06:02:23Z", "mode": "train", "global_step": 3066, "epoch": 0.30798593671521846, "loss": -0.1565, "grad_norm": 1.5562937259674072, "learning_rate": 7.121212121212122e-07, "num_tokens": 5743456.0, "completions/mean_length": 117.125, "completions/min_length": 56.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 60.71428680419922, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.8694585561752319, "rewards/meter/std": 0.33506831526756287, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9848867654800415, "rewards/repeat_soft/std": 0.014160656370222569, "rewards/judge_quality/mean": 0.39249998331069946, "rewards/judge_quality/std": 0.13905291259288788, "rewards/total_composite/mean": 0.7220953106880188, "rewards/total_composite/std": 0.2918062210083008, "reward": 0.7220953106880188, "reward_std": 0.2918062210083008, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09926391392946243, "sampling/sampling_logp_difference/max": 2.2996597290039062, "sampling/importance_sampling_ratio/min": 0.10029296576976776, "sampling/importance_sampling_ratio/mean": 0.9962590336799622, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3473818451166153, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08634534943848848, "clip_ratio/high_max": 0.08634534943848848, "clip_ratio/region_mean": 0.08634534943848848, "reward_total_mean": 0.7220953106880188, "reward_meter_mean": 0.8694585561752319, "reward_meter_std": 0.33506831526756287, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9848867654800415, "reward_repeat_soft_std": 0.014160656370222569, "reward_judge_quality_mean": 0.39249998331069946, "reward_judge_quality_std": 0.13905291259288788, "reward_total_composite_mean": 0.7220953106880188, "reward_total_composite_std": 0.2918062210083008} {"timestamp_utc": "2026-04-13T06:02:30Z", "mode": "train", "global_step": 3067, "epoch": 0.30808638874937216, "loss": 0.001, "grad_norm": 5.828978538513184, "learning_rate": 7.090909090909092e-07, "num_tokens": 5745923.0, "completions/mean_length": 133.375, "completions/min_length": 122.0, "completions/max_length": 148.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 133.375, "completions/min_terminated_length": 122.0, "completions/max_terminated_length": 148.0, "rewards/meter/mean": 0.9836386442184448, "rewards/meter/std": 0.013474264182150364, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8682862520217896, "rewards/repeat_soft/std": 0.0366973802447319, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.8103410005569458, "rewards/total_composite/std": 0.043476346880197525, "reward": 0.8103410005569458, "reward_std": 0.043476346880197525, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09245986491441727, "sampling/sampling_logp_difference/max": 3.5308737754821777, "sampling/importance_sampling_ratio/min": 0.029279321432113647, "sampling/importance_sampling_ratio/mean": 0.9992664456367493, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.365324754267931, "clip_ratio/low_mean": 0.0600816416554153, "clip_ratio/low_min": 0.0600816416554153, "clip_ratio/high_mean": 0.02387581579387188, "clip_ratio/high_max": 0.02387581579387188, "clip_ratio/region_mean": 0.08395745744928718, "reward_total_mean": 0.8103410005569458, "reward_meter_mean": 0.9836386442184448, "reward_meter_std": 0.013474264182150364, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8682862520217896, "reward_repeat_soft_std": 0.0366973802447319, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.8103410005569458, "reward_total_composite_std": 0.043476346880197525} {"timestamp_utc": "2026-04-13T06:02:37Z", "mode": "train", "global_step": 3068, "epoch": 0.3081868407835259, "loss": 0.0288, "grad_norm": 10.473501205444336, "learning_rate": 7.060606060606061e-07, "num_tokens": 5747601.0, "completions/mean_length": 51.75, "completions/min_length": 46.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.75, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9860801100730896, "rewards/meter/std": 0.0035912792664021254, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9767149090766907, "rewards/repeat_soft/std": 0.03255269303917885, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.838407576084137, "rewards/total_composite/std": 0.0535350926220417, "reward": 0.838407576084137, "reward_std": 0.0535350926220417, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10049033164978027, "sampling/sampling_logp_difference/max": 4.786960601806641, "sampling/importance_sampling_ratio/min": 0.008337760344147682, "sampling/importance_sampling_ratio/mean": 0.9886353611946106, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36582693085074425, "clip_ratio/low_mean": 0.05775050725787878, "clip_ratio/low_min": 0.05775050725787878, "clip_ratio/high_mean": 0.007075471803545952, "clip_ratio/high_max": 0.007075471803545952, "clip_ratio/region_mean": 0.06482597906142473, "reward_total_mean": 0.838407576084137, "reward_meter_mean": 0.9860801100730896, "reward_meter_std": 0.0035912792664021254, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9767149090766907, "reward_repeat_soft_std": 0.03255269303917885, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.838407576084137, "reward_total_composite_std": 0.0535350926220417} {"timestamp_utc": "2026-04-13T06:02:43Z", "mode": "train", "global_step": 3069, "epoch": 0.3082872928176796, "loss": 0.0473, "grad_norm": 25.08551025390625, "learning_rate": 7.03030303030303e-07, "num_tokens": 5749333.0, "completions/mean_length": 55.5, "completions/min_length": 51.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.5, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.7557427883148193, "rewards/meter/std": 0.36802569031715393, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9885768890380859, "rewards/repeat_soft/std": 0.00953869428485632, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.8086919784545898, "rewards/total_composite/std": 0.1809949427843094, "reward": 0.8086919784545898, "reward_std": 0.1809949427843094, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12547332048416138, "sampling/sampling_logp_difference/max": 1.632847785949707, "sampling/importance_sampling_ratio/min": 0.19537240266799927, "sampling/importance_sampling_ratio/mean": 0.9912445545196533, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5716092698276043, "clip_ratio/low_mean": 0.03231790475547314, "clip_ratio/low_min": 0.03231790475547314, "clip_ratio/high_mean": 0.05588845815509558, "clip_ratio/high_max": 0.05588845815509558, "clip_ratio/region_mean": 0.08820636291056871, "reward_total_mean": 0.8086919784545898, "reward_meter_mean": 0.7557427883148193, "reward_meter_std": 0.36802569031715393, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9885768890380859, "reward_repeat_soft_std": 0.00953869428485632, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.8086919784545898, "reward_total_composite_std": 0.1809949427843094} {"timestamp_utc": "2026-04-13T06:02:50Z", "mode": "train", "global_step": 3070, "epoch": 0.30838774485183323, "loss": -0.0208, "grad_norm": 10.695107460021973, "learning_rate": 7.000000000000001e-07, "num_tokens": 5750791.0, "completions/mean_length": 42.25, "completions/min_length": 37.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.25, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.6151335835456848, "rewards/meter/std": 0.43041181564331055, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9370051622390747, "rewards/repeat_soft/std": 0.054289545863866806, "rewards/judge_quality/mean": 0.4012500047683716, "rewards/judge_quality/std": 0.22793717682361603, "rewards/total_composite/mean": 0.640885591506958, "rewards/total_composite/std": 0.15207602083683014, "reward": 0.640885591506958, "reward_std": 0.15207602083683014, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07856753468513489, "sampling/sampling_logp_difference/max": 1.2095181941986084, "sampling/importance_sampling_ratio/min": 0.2983410060405731, "sampling/importance_sampling_ratio/mean": 0.9880566000938416, "sampling/importance_sampling_ratio/max": 1.8267732858657837, "entropy": 0.35338742658495903, "clip_ratio/low_mean": 0.015802159206941724, "clip_ratio/low_min": 0.015802159206941724, "clip_ratio/high_mean": 0.06838883273303509, "clip_ratio/high_max": 0.06838883273303509, "clip_ratio/region_mean": 0.08419099193997681, "reward_total_mean": 0.640885591506958, "reward_meter_mean": 0.6151335835456848, "reward_meter_std": 0.43041181564331055, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9370051622390747, "reward_repeat_soft_std": 0.054289545863866806, "reward_judge_quality_mean": 0.4012500047683716, "reward_judge_quality_std": 0.22793717682361603, "reward_total_composite_mean": 0.640885591506958, "reward_total_composite_std": 0.15207602083683014} {"timestamp_utc": "2026-04-13T06:02:57Z", "mode": "train", "global_step": 3071, "epoch": 0.30848819688598694, "loss": -0.0019, "grad_norm": 5.3360490798950195, "learning_rate": 6.969696969696971e-07, "num_tokens": 5753340.0, "completions/mean_length": 134.625, "completions/min_length": 123.0, "completions/max_length": 141.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 134.625, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 141.0, "rewards/meter/mean": 0.9899435639381409, "rewards/meter/std": 0.01261954102665186, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9028908014297485, "rewards/repeat_soft/std": 0.04720265790820122, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.7815761566162109, "rewards/total_composite/std": 0.03237314149737358, "reward": 0.7815761566162109, "reward_std": 0.03237314894795418, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07496866583824158, "sampling/sampling_logp_difference/max": 3.4019370079040527, "sampling/importance_sampling_ratio/min": 0.033308688551187515, "sampling/importance_sampling_ratio/mean": 1.0038471221923828, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.35843903943896294, "clip_ratio/low_mean": 0.026669048937037587, "clip_ratio/low_min": 0.026669048937037587, "clip_ratio/high_mean": 0.035919756162911654, "clip_ratio/high_max": 0.035919756162911654, "clip_ratio/region_mean": 0.06258880509994924, "reward_total_mean": 0.7815761566162109, "reward_meter_mean": 0.9899435639381409, "reward_meter_std": 0.01261954102665186, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9028908014297485, "reward_repeat_soft_std": 0.04720265790820122, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.7815761566162109, "reward_total_composite_std": 0.03237314149737358} {"timestamp_utc": "2026-04-13T06:03:04Z", "mode": "train", "global_step": 3072, "epoch": 0.30858864892014065, "loss": 0.0191, "grad_norm": 9.979864120483398, "learning_rate": 6.939393939393941e-07, "num_tokens": 5755195.0, "completions/mean_length": 60.875, "completions/min_length": 56.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9579508900642395, "rewards/meter/std": 0.05544468015432358, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9914168119430542, "rewards/repeat_soft/std": 0.011365259997546673, "rewards/judge_quality/mean": 0.6187499761581421, "rewards/judge_quality/std": 0.24976776540279388, "rewards/total_composite/mean": 0.8658446073532104, "rewards/total_composite/std": 0.08568636327981949, "reward": 0.8658446073532104, "reward_std": 0.0856863409280777, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10189951956272125, "sampling/sampling_logp_difference/max": 3.5550150871276855, "sampling/importance_sampling_ratio/min": 0.02858094498515129, "sampling/importance_sampling_ratio/mean": 1.0029480457305908, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.46703245863318443, "clip_ratio/low_mean": 0.05323939025402069, "clip_ratio/low_min": 0.05323939025402069, "clip_ratio/high_mean": 0.03528005536645651, "clip_ratio/high_max": 0.03528005536645651, "clip_ratio/region_mean": 0.0885194456204772, "reward_total_mean": 0.8658446073532104, "reward_meter_mean": 0.9579508900642395, "reward_meter_std": 0.05544468015432358, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9914168119430542, "reward_repeat_soft_std": 0.011365259997546673, "reward_judge_quality_mean": 0.6187499761581421, "reward_judge_quality_std": 0.24976776540279388, "reward_total_composite_mean": 0.8658446073532104, "reward_total_composite_std": 0.08568636327981949} {"timestamp_utc": "2026-04-13T06:03:11Z", "mode": "train", "global_step": 3073, "epoch": 0.3086891009542943, "loss": 0.0305, "grad_norm": 7.048933506011963, "learning_rate": 6.90909090909091e-07, "num_tokens": 5757289.0, "completions/mean_length": 107.75, "completions/min_length": 99.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.75, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.8926213383674622, "rewards/meter/std": 0.28284361958503723, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9133425354957581, "rewards/repeat_soft/std": 0.04915476590394974, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7690138816833496, "rewards/total_composite/std": 0.12509866058826447, "reward": 0.7690138816833496, "reward_std": 0.12509866058826447, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0930352434515953, "sampling/sampling_logp_difference/max": 1.3487974405288696, "sampling/importance_sampling_ratio/min": 0.26604753732681274, "sampling/importance_sampling_ratio/mean": 1.0069390535354614, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4254991225898266, "clip_ratio/low_mean": 0.018318966031074524, "clip_ratio/low_min": 0.018318966031074524, "clip_ratio/high_mean": 0.08510575164109468, "clip_ratio/high_max": 0.08510575164109468, "clip_ratio/region_mean": 0.10342471767216921, "reward_total_mean": 0.7690138816833496, "reward_meter_mean": 0.8926213383674622, "reward_meter_std": 0.28284361958503723, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9133425354957581, "reward_repeat_soft_std": 0.04915476590394974, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7690138816833496, "reward_total_composite_std": 0.12509866058826447} {"timestamp_utc": "2026-04-13T06:03:18Z", "mode": "train", "global_step": 3074, "epoch": 0.308789552988448, "loss": 0.0208, "grad_norm": 6.438065528869629, "learning_rate": 6.878787878787879e-07, "num_tokens": 5759856.0, "completions/mean_length": 129.875, "completions/min_length": 118.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 129.875, "completions/min_terminated_length": 118.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.9916092157363892, "rewards/meter/std": 0.002416644711047411, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7483363151550293, "rewards/repeat_soft/std": 0.11952370405197144, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7970578074455261, "rewards/total_composite/std": 0.01199943944811821, "reward": 0.7970578074455261, "reward_std": 0.011999450623989105, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0672622099518776, "sampling/sampling_logp_difference/max": 2.493488311767578, "sampling/importance_sampling_ratio/min": 0.0826212540268898, "sampling/importance_sampling_ratio/mean": 1.002930998802185, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2797844596207142, "clip_ratio/low_mean": 0.01900999597273767, "clip_ratio/low_min": 0.01900999597273767, "clip_ratio/high_mean": 0.03289586445316672, "clip_ratio/high_max": 0.03289586445316672, "clip_ratio/region_mean": 0.05190586042590439, "reward_total_mean": 0.7970578074455261, "reward_meter_mean": 0.9916092157363892, "reward_meter_std": 0.002416644711047411, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7483363151550293, "reward_repeat_soft_std": 0.11952370405197144, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7970578074455261, "reward_total_composite_std": 0.01199943944811821} {"timestamp_utc": "2026-04-13T06:03:25Z", "mode": "train", "global_step": 3075, "epoch": 0.3088900050226017, "loss": 0.0333, "grad_norm": 8.071162223815918, "learning_rate": 6.848484848484849e-07, "num_tokens": 5761841.0, "completions/mean_length": 87.125, "completions/min_length": 85.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.125, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9859744310379028, "rewards/meter/std": 0.004668372217565775, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9658116698265076, "rewards/repeat_soft/std": 0.03591784089803696, "rewards/judge_quality/mean": 0.7200000286102295, "rewards/judge_quality/std": 0.20701968669891357, "rewards/total_composite/mean": 0.9062696695327759, "rewards/total_composite/std": 0.059438735246658325, "reward": 0.9062696695327759, "reward_std": 0.05943873152136803, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07456713914871216, "sampling/sampling_logp_difference/max": 1.1077518463134766, "sampling/importance_sampling_ratio/min": 0.3303007185459137, "sampling/importance_sampling_ratio/mean": 1.0078984498977661, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4186944290995598, "clip_ratio/low_mean": 0.022222222294658422, "clip_ratio/low_min": 0.022222222294658422, "clip_ratio/high_mean": 0.06379767367616296, "clip_ratio/high_max": 0.06379767367616296, "clip_ratio/region_mean": 0.08601989597082138, "reward_total_mean": 0.9062696695327759, "reward_meter_mean": 0.9859744310379028, "reward_meter_std": 0.004668372217565775, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9658116698265076, "reward_repeat_soft_std": 0.03591784089803696, "reward_judge_quality_mean": 0.7200000286102295, "reward_judge_quality_std": 0.20701968669891357, "reward_total_composite_mean": 0.9062696695327759, "reward_total_composite_std": 0.059438735246658325} {"timestamp_utc": "2026-04-13T06:03:32Z", "mode": "train", "global_step": 3076, "epoch": 0.3089904570567554, "loss": 0.0166, "grad_norm": 5.150601863861084, "learning_rate": 6.818181818181818e-07, "num_tokens": 5764448.0, "completions/mean_length": 145.875, "completions/min_length": 141.0, "completions/max_length": 149.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 145.875, "completions/min_terminated_length": 141.0, "completions/max_terminated_length": 149.0, "rewards/meter/mean": 0.9865021705627441, "rewards/meter/std": 0.004395131021738052, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8216436505317688, "rewards/repeat_soft/std": 0.07443635910749435, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.12631450593471527, "rewards/total_composite/mean": 0.7754653692245483, "rewards/total_composite/std": 0.038631632924079895, "reward": 0.7754653692245483, "reward_std": 0.03863164409995079, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06353172659873962, "sampling/sampling_logp_difference/max": 1.6569032669067383, "sampling/importance_sampling_ratio/min": 0.19072870910167694, "sampling/importance_sampling_ratio/mean": 1.0015277862548828, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3011839520186186, "clip_ratio/low_mean": 0.02043224684894085, "clip_ratio/low_min": 0.02043224684894085, "clip_ratio/high_mean": 0.040138378739356995, "clip_ratio/high_max": 0.040138378739356995, "clip_ratio/region_mean": 0.060570625588297844, "reward_total_mean": 0.7754653692245483, "reward_meter_mean": 0.9865021705627441, "reward_meter_std": 0.004395131021738052, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8216436505317688, "reward_repeat_soft_std": 0.07443635910749435, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.12631450593471527, "reward_total_composite_mean": 0.7754653692245483, "reward_total_composite_std": 0.038631632924079895} {"timestamp_utc": "2026-04-13T06:03:39Z", "mode": "train", "global_step": 3077, "epoch": 0.3090909090909091, "loss": 0.052, "grad_norm": 8.09042739868164, "learning_rate": 6.78787878787879e-07, "num_tokens": 5766591.0, "completions/mean_length": 85.875, "completions/min_length": 77.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.875, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.9851846098899841, "rewards/meter/std": 0.007634002715349197, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9634048342704773, "rewards/repeat_soft/std": 0.01854957826435566, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.8029235601425171, "rewards/total_composite/std": 0.022176286205649376, "reward": 0.8029235601425171, "reward_std": 0.022176284343004227, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0990520566701889, "sampling/sampling_logp_difference/max": 1.4478211402893066, "sampling/importance_sampling_ratio/min": 0.23508194088935852, "sampling/importance_sampling_ratio/mean": 1.0035712718963623, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44220760464668274, "clip_ratio/low_mean": 0.02322488883510232, "clip_ratio/low_min": 0.02322488883510232, "clip_ratio/high_mean": 0.06331886001862586, "clip_ratio/high_max": 0.06331886001862586, "clip_ratio/region_mean": 0.08654374885372818, "reward_total_mean": 0.8029235601425171, "reward_meter_mean": 0.9851846098899841, "reward_meter_std": 0.007634002715349197, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9634048342704773, "reward_repeat_soft_std": 0.01854957826435566, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.8029235601425171, "reward_total_composite_std": 0.022176286205649376} {"timestamp_utc": "2026-04-13T06:03:46Z", "mode": "train", "global_step": 3078, "epoch": 0.3091913611250628, "loss": -0.0185, "grad_norm": 7.285684585571289, "learning_rate": 6.757575757575759e-07, "num_tokens": 5768973.0, "completions/mean_length": 115.75, "completions/min_length": 97.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.75, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.789310097694397, "rewards/meter/std": 0.09320956468582153, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9367285966873169, "rewards/repeat_soft/std": 0.03354889899492264, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.7286124229431152, "rewards/total_composite/std": 0.04265909269452095, "reward": 0.7286124229431152, "reward_std": 0.04265910014510155, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09601914882659912, "sampling/sampling_logp_difference/max": 2.622750759124756, "sampling/importance_sampling_ratio/min": 0.07260287553071976, "sampling/importance_sampling_ratio/mean": 0.9981330633163452, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.33441162109375, "clip_ratio/low_mean": 0.028539294842630625, "clip_ratio/low_min": 0.028539294842630625, "clip_ratio/high_mean": 0.06729705072939396, "clip_ratio/high_max": 0.06729705072939396, "clip_ratio/region_mean": 0.09583634557202458, "reward_total_mean": 0.7286124229431152, "reward_meter_mean": 0.789310097694397, "reward_meter_std": 0.09320956468582153, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9367285966873169, "reward_repeat_soft_std": 0.03354889899492264, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.7286124229431152, "reward_total_composite_std": 0.04265909269452095} {"timestamp_utc": "2026-04-13T06:03:58Z", "mode": "train", "global_step": 3079, "epoch": 0.3092918131592165, "loss": -0.1506, "grad_norm": 3.0242953300476074, "learning_rate": 6.727272727272728e-07, "num_tokens": 5771286.0, "completions/mean_length": 171.125, "completions/min_length": 116.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 122.42857360839844, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.8090323209762573, "rewards/meter/std": 0.3353683054447174, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9363648295402527, "rewards/repeat_soft/std": 0.026785895228385925, "rewards/judge_quality/mean": 0.3399999737739563, "rewards/judge_quality/std": 0.11747340112924576, "rewards/total_composite/mean": 0.7097010612487793, "rewards/total_composite/std": 0.1677362322807312, "reward": 0.7097010612487793, "reward_std": 0.1677362322807312, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08408427983522415, "sampling/sampling_logp_difference/max": 1.3533430099487305, "sampling/importance_sampling_ratio/min": 0.25837504863739014, "sampling/importance_sampling_ratio/mean": 1.0024542808532715, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31197330355644226, "clip_ratio/low_mean": 0.011278195306658745, "clip_ratio/low_min": 0.011278195306658745, "clip_ratio/high_mean": 0.058742071967571974, "clip_ratio/high_max": 0.058742071967571974, "clip_ratio/region_mean": 0.07002026727423072, "reward_total_mean": 0.7097010612487793, "reward_meter_mean": 0.8090323209762573, "reward_meter_std": 0.3353683054447174, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9363648295402527, "reward_repeat_soft_std": 0.026785895228385925, "reward_judge_quality_mean": 0.3399999737739563, "reward_judge_quality_std": 0.11747340112924576, "reward_total_composite_mean": 0.7097010612487793, "reward_total_composite_std": 0.1677362322807312} {"timestamp_utc": "2026-04-13T06:04:04Z", "mode": "train", "global_step": 3080, "epoch": 0.30939226519337015, "loss": -0.0152, "grad_norm": 11.691953659057617, "learning_rate": 6.696969696969698e-07, "num_tokens": 5772930.0, "completions/mean_length": 56.5, "completions/min_length": 51.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.5, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9902211427688599, "rewards/meter/std": 0.00952863972634077, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9857223033905029, "rewards/repeat_soft/std": 0.017040744423866272, "rewards/judge_quality/mean": 0.6424999833106995, "rewards/judge_quality/std": 0.2351139634847641, "rewards/total_composite/mean": 0.8869217038154602, "rewards/total_composite/std": 0.07266746461391449, "reward": 0.8869217038154602, "reward_std": 0.07266746461391449, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09294886142015457, "sampling/sampling_logp_difference/max": 1.4410853385925293, "sampling/importance_sampling_ratio/min": 0.23667074739933014, "sampling/importance_sampling_ratio/mean": 1.0078586339950562, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4520791247487068, "clip_ratio/low_mean": 0.06696659047156572, "clip_ratio/low_min": 0.06696659047156572, "clip_ratio/high_mean": 0.050499580800533295, "clip_ratio/high_max": 0.050499580800533295, "clip_ratio/region_mean": 0.11746617127209902, "reward_total_mean": 0.8869217038154602, "reward_meter_mean": 0.9902211427688599, "reward_meter_std": 0.00952863972634077, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9857223033905029, "reward_repeat_soft_std": 0.017040744423866272, "reward_judge_quality_mean": 0.6424999833106995, "reward_judge_quality_std": 0.2351139634847641, "reward_total_composite_mean": 0.8869217038154602, "reward_total_composite_std": 0.07266746461391449} {"timestamp_utc": "2026-04-13T06:04:10Z", "mode": "train", "global_step": 3081, "epoch": 0.30949271722752386, "loss": 0.0203, "grad_norm": 42.73087692260742, "learning_rate": 6.666666666666667e-07, "num_tokens": 5774543.0, "completions/mean_length": 47.625, "completions/min_length": 43.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.625, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.9255618453025818, "rewards/meter/std": 0.1816474050283432, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9215962886810303, "rewards/repeat_soft/std": 0.04104834794998169, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.7891625165939331, "rewards/total_composite/std": 0.0816275030374527, "reward": 0.7891625165939331, "reward_std": 0.0816274881362915, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09778644889593124, "sampling/sampling_logp_difference/max": 1.8514859676361084, "sampling/importance_sampling_ratio/min": 0.15700368583202362, "sampling/importance_sampling_ratio/mean": 1.0085833072662354, "sampling/importance_sampling_ratio/max": 1.9510786533355713, "entropy": 0.5184796638786793, "clip_ratio/low_mean": 0.0028409091755747795, "clip_ratio/low_min": 0.0028409091755747795, "clip_ratio/high_mean": 0.09789831470698118, "clip_ratio/high_max": 0.09789831470698118, "clip_ratio/region_mean": 0.10073922388255596, "reward_total_mean": 0.7891625165939331, "reward_meter_mean": 0.9255618453025818, "reward_meter_std": 0.1816474050283432, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9215962886810303, "reward_repeat_soft_std": 0.04104834794998169, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.7891625165939331, "reward_total_composite_std": 0.0816275030374527} {"timestamp_utc": "2026-04-13T06:04:22Z", "mode": "train", "global_step": 3082, "epoch": 0.30959316926167757, "loss": -0.1304, "grad_norm": 1.0314031839370728, "learning_rate": 6.636363636363636e-07, "num_tokens": 5776204.0, "completions/mean_length": 102.625, "completions/min_length": 43.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 44.142860412597656, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.8616480827331543, "rewards/meter/std": 0.30309802293777466, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8858416676521301, "rewards/repeat_soft/std": 0.07175854593515396, "rewards/judge_quality/mean": 0.26749998331069946, "rewards/judge_quality/std": 0.11671087145805359, "rewards/total_composite/mean": 0.6671397686004639, "rewards/total_composite/std": 0.2708699405193329, "reward": 0.6671397686004639, "reward_std": 0.2708699107170105, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05509642884135246, "sampling/sampling_logp_difference/max": 2.811518430709839, "sampling/importance_sampling_ratio/min": 0.06011364236474037, "sampling/importance_sampling_ratio/mean": 0.9968774914741516, "sampling/importance_sampling_ratio/max": 1.7178781032562256, "entropy": 0.1866202987730503, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.03972352878190577, "clip_ratio/high_max": 0.03972352878190577, "clip_ratio/region_mean": 0.03972352878190577, "reward_total_mean": 0.6671397686004639, "reward_meter_mean": 0.8616480827331543, "reward_meter_std": 0.30309802293777466, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8858416676521301, "reward_repeat_soft_std": 0.07175854593515396, "reward_judge_quality_mean": 0.26749998331069946, "reward_judge_quality_std": 0.11671087145805359, "reward_total_composite_mean": 0.6671397686004639, "reward_total_composite_std": 0.2708699405193329} {"timestamp_utc": "2026-04-13T06:04:28Z", "mode": "train", "global_step": 3083, "epoch": 0.3096936212958312, "loss": -0.0066, "grad_norm": 6.823232173919678, "learning_rate": 6.606060606060606e-07, "num_tokens": 5777976.0, "completions/mean_length": 69.5, "completions/min_length": 65.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.5, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9315026998519897, "rewards/meter/std": 0.07252227514982224, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8542488217353821, "rewards/repeat_soft/std": 0.06951781362295151, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7742260694503784, "rewards/total_composite/std": 0.03455724939703941, "reward": 0.7742260694503784, "reward_std": 0.034557241946458817, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06768452376127243, "sampling/sampling_logp_difference/max": 1.8914222717285156, "sampling/importance_sampling_ratio/min": 0.1508570909500122, "sampling/importance_sampling_ratio/mean": 1.0055797100067139, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3044958673417568, "clip_ratio/low_mean": 0.02239819010719657, "clip_ratio/low_min": 0.02239819010719657, "clip_ratio/high_mean": 0.044425095431506634, "clip_ratio/high_max": 0.044425095431506634, "clip_ratio/region_mean": 0.0668232855387032, "reward_total_mean": 0.7742260694503784, "reward_meter_mean": 0.9315026998519897, "reward_meter_std": 0.07252227514982224, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8542488217353821, "reward_repeat_soft_std": 0.06951781362295151, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7742260694503784, "reward_total_composite_std": 0.03455724939703941} {"timestamp_utc": "2026-04-13T06:04:35Z", "mode": "train", "global_step": 3084, "epoch": 0.30979407332998493, "loss": 0.0281, "grad_norm": 9.906598091125488, "learning_rate": 6.575757575757575e-07, "num_tokens": 5779647.0, "completions/mean_length": 59.875, "completions/min_length": 53.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.875, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.7166187167167664, "rewards/meter/std": 0.3853898048400879, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9471012353897095, "rewards/repeat_soft/std": 0.08057115972042084, "rewards/judge_quality/mean": 0.44624999165534973, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.7010635137557983, "rewards/total_composite/std": 0.17765507102012634, "reward": 0.7010635137557983, "reward_std": 0.17765507102012634, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0987834706902504, "sampling/sampling_logp_difference/max": 1.2055718898773193, "sampling/importance_sampling_ratio/min": 0.29952067136764526, "sampling/importance_sampling_ratio/mean": 1.0034987926483154, "sampling/importance_sampling_ratio/max": 1.8698740005493164, "entropy": 0.5342771224677563, "clip_ratio/low_mean": 0.02496626228094101, "clip_ratio/low_min": 0.02496626228094101, "clip_ratio/high_mean": 0.09568720497190952, "clip_ratio/high_max": 0.09568720497190952, "clip_ratio/region_mean": 0.12065346725285053, "reward_total_mean": 0.7010635137557983, "reward_meter_mean": 0.7166187167167664, "reward_meter_std": 0.3853898048400879, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9471012353897095, "reward_repeat_soft_std": 0.08057115972042084, "reward_judge_quality_mean": 0.44624999165534973, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.7010635137557983, "reward_total_composite_std": 0.17765507102012634} {"timestamp_utc": "2026-04-13T06:04:41Z", "mode": "train", "global_step": 3085, "epoch": 0.30989452536413864, "loss": 0.0831, "grad_norm": 10.889636993408203, "learning_rate": 6.545454545454547e-07, "num_tokens": 5781350.0, "completions/mean_length": 47.875, "completions/min_length": 39.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.875, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9735107421875, "rewards/meter/std": 0.021135374903678894, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9682822227478027, "rewards/repeat_soft/std": 0.027064528316259384, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.1940544992685318, "rewards/total_composite/mean": 0.8244080543518066, "rewards/total_composite/std": 0.06024119630455971, "reward": 0.8244080543518066, "reward_std": 0.0602412149310112, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1005353257060051, "sampling/sampling_logp_difference/max": 1.4414892196655273, "sampling/importance_sampling_ratio/min": 0.2365752011537552, "sampling/importance_sampling_ratio/mean": 1.0072871446609497, "sampling/importance_sampling_ratio/max": 1.8396446704864502, "entropy": 0.5528210401535034, "clip_ratio/low_mean": 0.05290488572791219, "clip_ratio/low_min": 0.05290488572791219, "clip_ratio/high_mean": 0.013605442363768816, "clip_ratio/high_max": 0.013605442363768816, "clip_ratio/region_mean": 0.066510328091681, "reward_total_mean": 0.8244080543518066, "reward_meter_mean": 0.9735107421875, "reward_meter_std": 0.021135374903678894, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9682822227478027, "reward_repeat_soft_std": 0.027064528316259384, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.1940544992685318, "reward_total_composite_mean": 0.8244080543518066, "reward_total_composite_std": 0.06024119630455971} {"timestamp_utc": "2026-04-13T06:04:47Z", "mode": "train", "global_step": 3086, "epoch": 0.3099949773982923, "loss": 0.0057, "grad_norm": 7.87661600112915, "learning_rate": 6.515151515151516e-07, "num_tokens": 5783052.0, "completions/mean_length": 56.75, "completions/min_length": 52.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.75, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9895950555801392, "rewards/meter/std": 0.010515482164919376, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.81368088722229, "rewards/repeat_soft/std": 0.09412098675966263, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.8139358758926392, "rewards/total_composite/std": 0.032643262296915054, "reward": 0.8139358758926392, "reward_std": 0.032643258571624756, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05753866210579872, "sampling/sampling_logp_difference/max": 0.9461120367050171, "sampling/importance_sampling_ratio/min": 0.38824760913848877, "sampling/importance_sampling_ratio/mean": 1.0154799222946167, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.32913435250520706, "clip_ratio/low_mean": 0.03938857279717922, "clip_ratio/low_min": 0.03938857279717922, "clip_ratio/high_mean": 0.0069444444961845875, "clip_ratio/high_max": 0.0069444444961845875, "clip_ratio/region_mean": 0.04633301729336381, "reward_total_mean": 0.8139358758926392, "reward_meter_mean": 0.9895950555801392, "reward_meter_std": 0.010515482164919376, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.81368088722229, "reward_repeat_soft_std": 0.09412098675966263, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.8139358758926392, "reward_total_composite_std": 0.032643262296915054} {"timestamp_utc": "2026-04-13T06:04:54Z", "mode": "train", "global_step": 3087, "epoch": 0.310095429432446, "loss": 0.0382, "grad_norm": 5.365631580352783, "learning_rate": 6.484848484848485e-07, "num_tokens": 5785396.0, "completions/mean_length": 111.0, "completions/min_length": 98.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.0, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.9909769892692566, "rewards/meter/std": 0.00389321381226182, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9604401588439941, "rewards/repeat_soft/std": 0.004026754759252071, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.8078586459159851, "rewards/total_composite/std": 0.02803451381623745, "reward": 0.8078586459159851, "reward_std": 0.028034524992108345, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0915830135345459, "sampling/sampling_logp_difference/max": 1.6850559711456299, "sampling/importance_sampling_ratio/min": 0.18543405830860138, "sampling/importance_sampling_ratio/mean": 1.001240849494934, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40845757722854614, "clip_ratio/low_mean": 0.009615384973585606, "clip_ratio/low_min": 0.009615384973585606, "clip_ratio/high_mean": 0.08097112365067005, "clip_ratio/high_max": 0.08097112365067005, "clip_ratio/region_mean": 0.09058650862425566, "reward_total_mean": 0.8078586459159851, "reward_meter_mean": 0.9909769892692566, "reward_meter_std": 0.00389321381226182, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9604401588439941, "reward_repeat_soft_std": 0.004026754759252071, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.8078586459159851, "reward_total_composite_std": 0.02803451381623745} {"timestamp_utc": "2026-04-13T06:05:00Z", "mode": "train", "global_step": 3088, "epoch": 0.3101958814665997, "loss": 0.0602, "grad_norm": 9.739012718200684, "learning_rate": 6.454545454545455e-07, "num_tokens": 5787105.0, "completions/mean_length": 50.625, "completions/min_length": 46.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.625, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9666648507118225, "rewards/meter/std": 0.04159614071249962, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9984562397003174, "rewards/repeat_soft/std": 0.003634644439443946, "rewards/judge_quality/mean": 0.59375, "rewards/judge_quality/std": 0.22398583590984344, "rewards/total_composite/mean": 0.8629698157310486, "rewards/total_composite/std": 0.07638230919837952, "reward": 0.8629698157310486, "reward_std": 0.07638231664896011, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0794394239783287, "sampling/sampling_logp_difference/max": 1.6646841764450073, "sampling/importance_sampling_ratio/min": 0.1892504096031189, "sampling/importance_sampling_ratio/mean": 0.9849066734313965, "sampling/importance_sampling_ratio/max": 1.9696766138076782, "entropy": 0.3073462024331093, "clip_ratio/low_mean": 0.040070064598694444, "clip_ratio/low_min": 0.040070064598694444, "clip_ratio/high_mean": 0.02350451098755002, "clip_ratio/high_max": 0.02350451098755002, "clip_ratio/region_mean": 0.06357457558624446, "reward_total_mean": 0.8629698157310486, "reward_meter_mean": 0.9666648507118225, "reward_meter_std": 0.04159614071249962, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9984562397003174, "reward_repeat_soft_std": 0.003634644439443946, "reward_judge_quality_mean": 0.59375, "reward_judge_quality_std": 0.22398583590984344, "reward_total_composite_mean": 0.8629698157310486, "reward_total_composite_std": 0.07638230919837952} {"timestamp_utc": "2026-04-13T06:05:07Z", "mode": "train", "global_step": 3089, "epoch": 0.31029633350075336, "loss": 0.0336, "grad_norm": 8.70956039428711, "learning_rate": 6.424242424242424e-07, "num_tokens": 5788761.0, "completions/mean_length": 57.0, "completions/min_length": 49.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.0, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9844430088996887, "rewards/meter/std": 0.00920199602842331, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9552140235900879, "rewards/repeat_soft/std": 0.035446420311927795, "rewards/judge_quality/mean": 0.44999998807907104, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8235207796096802, "rewards/total_composite/std": 0.005225022789090872, "reward": 0.8235207796096802, "reward_std": 0.005225030705332756, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10148558765649796, "sampling/sampling_logp_difference/max": 3.237790107727051, "sampling/importance_sampling_ratio/min": 0.039250537753105164, "sampling/importance_sampling_ratio/mean": 0.992186427116394, "sampling/importance_sampling_ratio/max": 1.7431141138076782, "entropy": 0.46908702701330185, "clip_ratio/low_mean": 0.05621598241850734, "clip_ratio/low_min": 0.05621598241850734, "clip_ratio/high_mean": 0.0537731908261776, "clip_ratio/high_max": 0.0537731908261776, "clip_ratio/region_mean": 0.10998917324468493, "reward_total_mean": 0.8235207796096802, "reward_meter_mean": 0.9844430088996887, "reward_meter_std": 0.00920199602842331, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9552140235900879, "reward_repeat_soft_std": 0.035446420311927795, "reward_judge_quality_mean": 0.44999998807907104, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8235207796096802, "reward_total_composite_std": 0.005225022789090872} {"timestamp_utc": "2026-04-13T06:05:14Z", "mode": "train", "global_step": 3090, "epoch": 0.31039678553490707, "loss": -0.0225, "grad_norm": 8.803093910217285, "learning_rate": 6.393939393939394e-07, "num_tokens": 5790554.0, "completions/mean_length": 57.125, "completions/min_length": 52.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.125, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.977512001991272, "rewards/meter/std": 0.020044323056936264, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9755935668945312, "rewards/repeat_soft/std": 0.023083893582224846, "rewards/judge_quality/mean": 0.44999998807907104, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8224397897720337, "rewards/total_composite/std": 0.010517973452806473, "reward": 0.8224397897720337, "reward_std": 0.01051798090338707, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0844256654381752, "sampling/sampling_logp_difference/max": 1.6369256973266602, "sampling/importance_sampling_ratio/min": 0.19457730650901794, "sampling/importance_sampling_ratio/mean": 0.9926048517227173, "sampling/importance_sampling_ratio/max": 1.67318594455719, "entropy": 0.38883984461426735, "clip_ratio/low_mean": 0.01878561219200492, "clip_ratio/low_min": 0.01878561219200492, "clip_ratio/high_mean": 0.0619459068402648, "clip_ratio/high_max": 0.0619459068402648, "clip_ratio/region_mean": 0.08073151903226972, "reward_total_mean": 0.8224397897720337, "reward_meter_mean": 0.977512001991272, "reward_meter_std": 0.020044323056936264, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9755935668945312, "reward_repeat_soft_std": 0.023083893582224846, "reward_judge_quality_mean": 0.44999998807907104, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8224397897720337, "reward_total_composite_std": 0.010517973452806473} {"timestamp_utc": "2026-04-13T06:05:21Z", "mode": "train", "global_step": 3091, "epoch": 0.3104972375690608, "loss": 0.0065, "grad_norm": 6.499277591705322, "learning_rate": 6.363636363636364e-07, "num_tokens": 5792765.0, "completions/mean_length": 90.375, "completions/min_length": 75.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.375, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.9774361848831177, "rewards/meter/std": 0.018627861514687538, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9568881988525391, "rewards/repeat_soft/std": 0.02445368468761444, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.823910117149353, "rewards/total_composite/std": 0.06016247346997261, "reward": 0.823910117149353, "reward_std": 0.060162488371133804, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09028123319149017, "sampling/sampling_logp_difference/max": 1.354330062866211, "sampling/importance_sampling_ratio/min": 0.2581201493740082, "sampling/importance_sampling_ratio/mean": 1.002707839012146, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4370978847146034, "clip_ratio/low_mean": 0.057562037371098995, "clip_ratio/low_min": 0.057562037371098995, "clip_ratio/high_mean": 0.01822916604578495, "clip_ratio/high_max": 0.01822916604578495, "clip_ratio/region_mean": 0.07579120341688395, "reward_total_mean": 0.823910117149353, "reward_meter_mean": 0.9774361848831177, "reward_meter_std": 0.018627861514687538, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9568881988525391, "reward_repeat_soft_std": 0.02445368468761444, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.823910117149353, "reward_total_composite_std": 0.06016247346997261} {"timestamp_utc": "2026-04-13T06:05:29Z", "mode": "train", "global_step": 3092, "epoch": 0.3105976896032145, "loss": -0.0128, "grad_norm": 13.611068725585938, "learning_rate": 6.333333333333334e-07, "num_tokens": 5795983.0, "completions/mean_length": 171.25, "completions/min_length": 136.0, "completions/max_length": 194.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 171.25, "completions/min_terminated_length": 136.0, "completions/max_terminated_length": 194.0, "rewards/meter/mean": 0.9948974847793579, "rewards/meter/std": 0.000646066153421998, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9236589670181274, "rewards/repeat_soft/std": 0.03185554966330528, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.8021947145462036, "rewards/total_composite/std": 0.02820361964404583, "reward": 0.8021947145462036, "reward_std": 0.02820361778140068, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08015190809965134, "sampling/sampling_logp_difference/max": 2.4227466583251953, "sampling/importance_sampling_ratio/min": 0.08867771178483963, "sampling/importance_sampling_ratio/mean": 1.0027854442596436, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36493632942438126, "clip_ratio/low_mean": 0.019782913848757744, "clip_ratio/low_min": 0.019782913848757744, "clip_ratio/high_mean": 0.05548172444105148, "clip_ratio/high_max": 0.05548172444105148, "clip_ratio/region_mean": 0.07526463828980923, "reward_total_mean": 0.8021947145462036, "reward_meter_mean": 0.9948974847793579, "reward_meter_std": 0.000646066153421998, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9236589670181274, "reward_repeat_soft_std": 0.03185554966330528, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.8021947145462036, "reward_total_composite_std": 0.02820361964404583} {"timestamp_utc": "2026-04-13T06:05:36Z", "mode": "train", "global_step": 3093, "epoch": 0.31069814163736814, "loss": 0.0166, "grad_norm": 7.94556188583374, "learning_rate": 6.303030303030304e-07, "num_tokens": 5798027.0, "completions/mean_length": 95.5, "completions/min_length": 83.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.5, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.9363009929656982, "rewards/meter/std": 0.10819363594055176, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9673541784286499, "rewards/repeat_soft/std": 0.019611142575740814, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.794070839881897, "rewards/total_composite/std": 0.04978923127055168, "reward": 0.794070839881897, "reward_std": 0.049789220094680786, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09770548343658447, "sampling/sampling_logp_difference/max": 1.9404361248016357, "sampling/importance_sampling_ratio/min": 0.14364129304885864, "sampling/importance_sampling_ratio/mean": 1.0068641901016235, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4468788914382458, "clip_ratio/low_mean": 0.014473684132099152, "clip_ratio/low_min": 0.014473684132099152, "clip_ratio/high_mean": 0.09505687560886145, "clip_ratio/high_max": 0.09505687560886145, "clip_ratio/region_mean": 0.1095305597409606, "reward_total_mean": 0.794070839881897, "reward_meter_mean": 0.9363009929656982, "reward_meter_std": 0.10819363594055176, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9673541784286499, "reward_repeat_soft_std": 0.019611142575740814, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.794070839881897, "reward_total_composite_std": 0.04978923127055168} {"timestamp_utc": "2026-04-13T06:05:42Z", "mode": "train", "global_step": 3094, "epoch": 0.31079859367152185, "loss": 0.0157, "grad_norm": 13.058426856994629, "learning_rate": 6.272727272727273e-07, "num_tokens": 5799664.0, "completions/mean_length": 37.625, "completions/min_length": 36.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.625, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.9703361392021179, "rewards/meter/std": 0.06665296107530594, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624544382095337, "rewards/repeat_soft/std": 0.02980038709938526, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8122717142105103, "rewards/total_composite/std": 0.0312277153134346, "reward": 0.8122717142105103, "reward_std": 0.031227711588144302, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06885404139757156, "sampling/sampling_logp_difference/max": 1.5549311637878418, "sampling/importance_sampling_ratio/min": 0.2112039178609848, "sampling/importance_sampling_ratio/mean": 1.0102605819702148, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3112759608775377, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/high_mean": 0.04011463467031717, "clip_ratio/high_max": 0.04011463467031717, "clip_ratio/region_mean": 0.04349301313050091, "reward_total_mean": 0.8122717142105103, "reward_meter_mean": 0.9703361392021179, "reward_meter_std": 0.06665296107530594, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624544382095337, "reward_repeat_soft_std": 0.02980038709938526, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8122717142105103, "reward_total_composite_std": 0.0312277153134346} {"timestamp_utc": "2026-04-13T06:05:49Z", "mode": "train", "global_step": 3095, "epoch": 0.31089904570567556, "loss": 0.0098, "grad_norm": 6.736450672149658, "learning_rate": 6.242424242424243e-07, "num_tokens": 5801469.0, "completions/mean_length": 68.625, "completions/min_length": 60.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.625, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.990399181842804, "rewards/meter/std": 0.005728308577090502, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9567740559577942, "rewards/repeat_soft/std": 0.04929469898343086, "rewards/judge_quality/mean": 0.6487500071525574, "rewards/judge_quality/std": 0.2507951855659485, "rewards/total_composite/mean": 0.8859820365905762, "rewards/total_composite/std": 0.07507769763469696, "reward": 0.8859820365905762, "reward_std": 0.07507769763469696, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0652025043964386, "sampling/sampling_logp_difference/max": 2.761874198913574, "sampling/importance_sampling_ratio/min": 0.06317325681447983, "sampling/importance_sampling_ratio/mean": 1.0158940553665161, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.38596154376864433, "clip_ratio/low_mean": 0.03953152149915695, "clip_ratio/low_min": 0.03953152149915695, "clip_ratio/high_mean": 0.030489131109789014, "clip_ratio/high_max": 0.030489131109789014, "clip_ratio/region_mean": 0.07002065260894597, "reward_total_mean": 0.8859820365905762, "reward_meter_mean": 0.990399181842804, "reward_meter_std": 0.005728308577090502, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9567740559577942, "reward_repeat_soft_std": 0.04929469898343086, "reward_judge_quality_mean": 0.6487500071525574, "reward_judge_quality_std": 0.2507951855659485, "reward_total_composite_mean": 0.8859820365905762, "reward_total_composite_std": 0.07507769763469696} {"timestamp_utc": "2026-04-13T06:06:00Z", "mode": "train", "global_step": 3096, "epoch": 0.3109994977398292, "loss": -0.1785, "grad_norm": 2.9638798236846924, "learning_rate": 6.212121212121212e-07, "num_tokens": 5803713.0, "completions/mean_length": 147.5, "completions/min_length": 86.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 95.42857360839844, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.6958595514297485, "rewards/meter/std": 0.3146068751811981, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.2828427255153656, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9855765700340271, "rewards/repeat_soft/std": 0.01080237701535225, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.2962353229522705, "rewards/total_composite/mean": 0.7096555233001709, "rewards/total_composite/std": 0.30299967527389526, "reward": 0.7096555233001709, "reward_std": 0.30299967527389526, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11143883317708969, "sampling/sampling_logp_difference/max": 2.93729305267334, "sampling/importance_sampling_ratio/min": 0.0530090257525444, "sampling/importance_sampling_ratio/mean": 1.0005522966384888, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.384110689163208, "clip_ratio/low_mean": 0.007575757801532745, "clip_ratio/low_min": 0.007575757801532745, "clip_ratio/high_mean": 0.0765702398493886, "clip_ratio/high_max": 0.0765702398493886, "clip_ratio/region_mean": 0.08414599765092134, "reward_total_mean": 0.7096555233001709, "reward_meter_mean": 0.6958595514297485, "reward_meter_std": 0.3146068751811981, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.2828427255153656, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9855765700340271, "reward_repeat_soft_std": 0.01080237701535225, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.2962353229522705, "reward_total_composite_mean": 0.7096555233001709, "reward_total_composite_std": 0.30299967527389526} {"timestamp_utc": "2026-04-13T06:06:12Z", "mode": "train", "global_step": 3097, "epoch": 0.3110999497739829, "loss": -0.151, "grad_norm": 2.0570764541625977, "learning_rate": 6.181818181818182e-07, "num_tokens": 5805435.0, "completions/mean_length": 114.25, "completions/min_length": 52.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 57.42857360839844, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9772098064422607, "rewards/meter/std": 0.03223555162549019, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9577066898345947, "rewards/repeat_soft/std": 0.027248388156294823, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.2876474857330322, "rewards/total_composite/mean": 0.7538259029388428, "rewards/total_composite/std": 0.31143155694007874, "reward": 0.7538259029388428, "reward_std": 0.31143152713775635, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11452040076255798, "sampling/sampling_logp_difference/max": 2.327181100845337, "sampling/importance_sampling_ratio/min": 0.09757039695978165, "sampling/importance_sampling_ratio/mean": 0.987808883190155, "sampling/importance_sampling_ratio/max": 1.9390225410461426, "entropy": 0.3957183435559273, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10879812017083168, "clip_ratio/high_max": 0.10879812017083168, "clip_ratio/region_mean": 0.10879812017083168, "reward_total_mean": 0.7538259029388428, "reward_meter_mean": 0.9772098064422607, "reward_meter_std": 0.03223555162549019, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9577066898345947, "reward_repeat_soft_std": 0.027248388156294823, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.2876474857330322, "reward_total_composite_mean": 0.7538259029388428, "reward_total_composite_std": 0.31143155694007874} {"timestamp_utc": "2026-04-13T06:06:18Z", "mode": "train", "global_step": 3098, "epoch": 0.31120040180813663, "loss": 0.0255, "grad_norm": 9.991089820861816, "learning_rate": 6.151515151515152e-07, "num_tokens": 5807236.0, "completions/mean_length": 69.125, "completions/min_length": 63.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.125, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9901505708694458, "rewards/meter/std": 0.004162302240729332, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9459233283996582, "rewards/repeat_soft/std": 0.037583742290735245, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.809785008430481, "rewards/total_composite/std": 0.0177072174847126, "reward": 0.809785008430481, "reward_std": 0.017707211896777153, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0817512571811676, "sampling/sampling_logp_difference/max": 2.413393497467041, "sampling/importance_sampling_ratio/min": 0.08951102197170258, "sampling/importance_sampling_ratio/mean": 0.9954307675361633, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3514958582818508, "clip_ratio/low_mean": 0.01056338008493185, "clip_ratio/low_min": 0.01056338008493185, "clip_ratio/high_mean": 0.0745904438663274, "clip_ratio/high_max": 0.0745904438663274, "clip_ratio/region_mean": 0.08515382395125926, "reward_total_mean": 0.809785008430481, "reward_meter_mean": 0.9901505708694458, "reward_meter_std": 0.004162302240729332, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9459233283996582, "reward_repeat_soft_std": 0.037583742290735245, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.809785008430481, "reward_total_composite_std": 0.0177072174847126} {"timestamp_utc": "2026-04-13T06:06:25Z", "mode": "train", "global_step": 3099, "epoch": 0.3113008538422903, "loss": 0.0464, "grad_norm": 9.40096664428711, "learning_rate": 6.121212121212121e-07, "num_tokens": 5809007.0, "completions/mean_length": 55.375, "completions/min_length": 48.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.375, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.964889407157898, "rewards/meter/std": 0.02613256126642227, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.990748405456543, "rewards/repeat_soft/std": 0.013280103914439678, "rewards/judge_quality/mean": 0.59375, "rewards/judge_quality/std": 0.22398583590984344, "rewards/total_composite/mean": 0.8614000082015991, "rewards/total_composite/std": 0.059494681656360626, "reward": 0.8614000082015991, "reward_std": 0.05949468910694122, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08663446456193924, "sampling/sampling_logp_difference/max": 1.1597249507904053, "sampling/importance_sampling_ratio/min": 0.3135724365711212, "sampling/importance_sampling_ratio/mean": 0.9985917806625366, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.463082704693079, "clip_ratio/low_mean": 0.04607483837753534, "clip_ratio/low_min": 0.04607483837753534, "clip_ratio/high_mean": 0.04297916032373905, "clip_ratio/high_max": 0.04297916032373905, "clip_ratio/region_mean": 0.0890539987012744, "reward_total_mean": 0.8614000082015991, "reward_meter_mean": 0.964889407157898, "reward_meter_std": 0.02613256126642227, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.990748405456543, "reward_repeat_soft_std": 0.013280103914439678, "reward_judge_quality_mean": 0.59375, "reward_judge_quality_std": 0.22398583590984344, "reward_total_composite_mean": 0.8614000082015991, "reward_total_composite_std": 0.059494681656360626} {"timestamp_utc": "2026-04-13T06:06:36Z", "mode": "train", "global_step": 3100, "epoch": 0.311401305876444, "loss": -0.0788, "grad_norm": 1.16935396194458, "learning_rate": 6.090909090909092e-07, "num_tokens": 5810495.0, "completions/mean_length": 151.0, "completions/min_length": 26.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 30.666667938232422, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.8612039089202881, "rewards/meter/std": 0.3366475999355316, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9541351795196533, "rewards/repeat_soft/std": 0.01282087154686451, "rewards/judge_quality/mean": 0.39500001072883606, "rewards/judge_quality/std": 0.283095121383667, "rewards/total_composite/mean": 0.6316143274307251, "rewards/total_composite/std": 0.3932740092277527, "reward": 0.6316143274307251, "reward_std": 0.3932739794254303, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09439637511968613, "sampling/sampling_logp_difference/max": 1.2195978164672852, "sampling/importance_sampling_ratio/min": 0.2953489124774933, "sampling/importance_sampling_ratio/mean": 1.0036258697509766, "sampling/importance_sampling_ratio/max": 1.8153910636901855, "entropy": 0.4194312207400799, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06553576746955514, "clip_ratio/high_max": 0.06553576746955514, "clip_ratio/region_mean": 0.06553576746955514, "reward_total_mean": 0.6316143274307251, "reward_meter_mean": 0.8612039089202881, "reward_meter_std": 0.3366475999355316, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9541351795196533, "reward_repeat_soft_std": 0.01282087154686451, "reward_judge_quality_mean": 0.39500001072883606, "reward_judge_quality_std": 0.283095121383667, "reward_total_composite_mean": 0.6316143274307251, "reward_total_composite_std": 0.3932740092277527} {"timestamp_utc": "2026-04-13T06:07:30Z", "mode": "eval", "global_step": 3100, "epoch": 0.311401305876444, "eval_loss": NaN, "eval_runtime": 53.8128, "eval_samples_per_second": 1.487, "eval_steps_per_second": 0.186, "eval_num_tokens": 5810495.0, "eval_completions/mean_length": 104.875, "eval_completions/min_length": 42.1, "eval_completions/max_length": 227.9, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 94.79821548461913, "eval_completions/min_terminated_length": 42.1, "eval_completions/max_terminated_length": 164.5, "eval_rewards/meter/mean": 0.9205826997756958, "eval_rewards/meter/std": 0.12805313621647657, "eval_rewards/count_adherence/mean": 0.9912500023841858, "eval_rewards/count_adherence/std": 0.024748736619949342, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.07071067690849304, "eval_rewards/repeat_soft/mean": 0.9113658428192138, "eval_rewards/repeat_soft/std": 0.07910711728036404, "eval_rewards/judge_quality/mean": 0.4437499940395355, "eval_rewards/judge_quality/std": 0.1451612905599177, "eval_rewards/total_composite/mean": 0.7749486446380616, "eval_rewards/total_composite/std": 0.11277823783457279, "eval_reward": 0.7749486446380616, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.03821708150207996, "eval_sampling/sampling_logp_difference/max": 0.9683253765106201, "eval_sampling/importance_sampling_ratio/min": 0.38764134645462034, "eval_sampling/importance_sampling_ratio/mean": 1.0066736578941344, "eval_sampling/importance_sampling_ratio/max": 1.3268948912620544, "eval_entropy": 0.3663712054491043, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7749486446380616, "eval_reward_meter_mean": 0.9205826997756958, "eval_reward_meter_std": 0.12805313621647657, "eval_reward_count_adherence_mean": 0.9912500023841858, "eval_reward_count_adherence_std": 0.024748736619949342, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.07071067690849304, "eval_reward_repeat_soft_mean": 0.9113658428192138, "eval_reward_repeat_soft_std": 0.07910711728036404, "eval_reward_judge_quality_mean": 0.4437499940395355, "eval_reward_judge_quality_std": 0.1451612905599177, "eval_reward_total_composite_mean": 0.7749486446380616, "eval_reward_total_composite_std": 0.11277823783457279} {"timestamp_utc": "2026-04-13T06:07:43Z", "mode": "train", "global_step": 3101, "epoch": 0.3115017579105977, "loss": 0.0282, "grad_norm": 6.122463703155518, "learning_rate": 6.060606060606061e-07, "num_tokens": 5812830.0, "completions/mean_length": 113.875, "completions/min_length": 107.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.875, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.9279730319976807, "rewards/meter/std": 0.16897909343242645, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9416385293006897, "rewards/repeat_soft/std": 0.03500296548008919, "rewards/judge_quality/mean": 0.5774999856948853, "rewards/judge_quality/std": 0.1609569787979126, "rewards/total_composite/mean": 0.8350017070770264, "rewards/total_composite/std": 0.10595095157623291, "reward": 0.8350017070770264, "reward_std": 0.10595094412565231, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07536301016807556, "sampling/sampling_logp_difference/max": 1.587273120880127, "sampling/importance_sampling_ratio/min": 0.20448245108127594, "sampling/importance_sampling_ratio/mean": 1.0013148784637451, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3822469040751457, "clip_ratio/low_mean": 0.04189633158966899, "clip_ratio/low_min": 0.04189633158966899, "clip_ratio/high_mean": 0.04032361879944801, "clip_ratio/high_max": 0.04032361879944801, "clip_ratio/region_mean": 0.082219950389117, "reward_total_mean": 0.8350017070770264, "reward_meter_mean": 0.9279730319976807, "reward_meter_std": 0.16897909343242645, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9416385293006897, "reward_repeat_soft_std": 0.03500296548008919, "reward_judge_quality_mean": 0.5774999856948853, "reward_judge_quality_std": 0.1609569787979126, "reward_total_composite_mean": 0.8350017070770264, "reward_total_composite_std": 0.10595095157623291} {"timestamp_utc": "2026-04-13T06:07:50Z", "mode": "train", "global_step": 3102, "epoch": 0.3116022099447514, "loss": 0.0478, "grad_norm": 7.018666744232178, "learning_rate": 6.03030303030303e-07, "num_tokens": 5814448.0, "completions/mean_length": 56.25, "completions/min_length": 53.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.25, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.945326030254364, "rewards/meter/std": 0.10452602803707123, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.905782163143158, "rewards/repeat_soft/std": 0.032895177602767944, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.8857249021530151, "rewards/total_composite/std": 0.07775601744651794, "reward": 0.8857249021530151, "reward_std": 0.07775601744651794, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07592600584030151, "sampling/sampling_logp_difference/max": 2.6686744689941406, "sampling/importance_sampling_ratio/min": 0.06934408098459244, "sampling/importance_sampling_ratio/mean": 1.0022361278533936, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.32013357616961, "clip_ratio/low_mean": 0.021709579043090343, "clip_ratio/low_min": 0.021709579043090343, "clip_ratio/high_mean": 0.02715097414329648, "clip_ratio/high_max": 0.02715097414329648, "clip_ratio/region_mean": 0.048860553186386824, "reward_total_mean": 0.8857249021530151, "reward_meter_mean": 0.945326030254364, "reward_meter_std": 0.10452602803707123, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.905782163143158, "reward_repeat_soft_std": 0.032895177602767944, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.8857249021530151, "reward_total_composite_std": 0.07775601744651794} {"timestamp_utc": "2026-04-13T06:07:55Z", "mode": "train", "global_step": 3103, "epoch": 0.31170266197890506, "loss": -0.0876, "grad_norm": 27.089813232421875, "learning_rate": 6.000000000000001e-07, "num_tokens": 5815803.0, "completions/mean_length": 19.375, "completions/min_length": 14.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.375, "completions/min_terminated_length": 14.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.7227749228477478, "rewards/meter/std": 0.4252164661884308, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9436079263687134, "rewards/repeat_soft/std": 0.053434766829013824, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.6701095104217529, "rewards/total_composite/std": 0.19269268214702606, "reward": 0.6701095104217529, "reward_std": 0.19269268214702606, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13997982442378998, "sampling/sampling_logp_difference/max": 3.6267507076263428, "sampling/importance_sampling_ratio/min": 0.02660248428583145, "sampling/importance_sampling_ratio/mean": 0.9844956994056702, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5855350978672504, "clip_ratio/low_mean": 0.02281746082007885, "clip_ratio/low_min": 0.02281746082007885, "clip_ratio/high_mean": 0.09790194453671575, "clip_ratio/high_max": 0.09790194453671575, "clip_ratio/region_mean": 0.1207194053567946, "reward_total_mean": 0.6701095104217529, "reward_meter_mean": 0.7227749228477478, "reward_meter_std": 0.4252164661884308, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9436079263687134, "reward_repeat_soft_std": 0.053434766829013824, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.6701095104217529, "reward_total_composite_std": 0.19269268214702606} {"timestamp_utc": "2026-04-13T06:08:02Z", "mode": "train", "global_step": 3104, "epoch": 0.31180311401305877, "loss": -0.0211, "grad_norm": 6.902459621429443, "learning_rate": 5.96969696969697e-07, "num_tokens": 5817515.0, "completions/mean_length": 57.0, "completions/min_length": 53.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.0, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9889857769012451, "rewards/meter/std": 0.0016639711102470756, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9660995602607727, "rewards/repeat_soft/std": 0.03163823485374451, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8187785148620605, "rewards/total_composite/std": 0.005233479663729668, "reward": 0.8187785148620605, "reward_std": 0.005233486648648977, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07573998719453812, "sampling/sampling_logp_difference/max": 2.349360466003418, "sampling/importance_sampling_ratio/min": 0.18393144011497498, "sampling/importance_sampling_ratio/mean": 1.0012197494506836, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.30078270472586155, "clip_ratio/low_mean": 0.03331019449979067, "clip_ratio/low_min": 0.03331019449979067, "clip_ratio/high_mean": 0.028282206505537033, "clip_ratio/high_max": 0.028282206505537033, "clip_ratio/region_mean": 0.0615924010053277, "reward_total_mean": 0.8187785148620605, "reward_meter_mean": 0.9889857769012451, "reward_meter_std": 0.0016639711102470756, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9660995602607727, "reward_repeat_soft_std": 0.03163823485374451, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8187785148620605, "reward_total_composite_std": 0.005233479663729668} {"timestamp_utc": "2026-04-13T06:08:08Z", "mode": "train", "global_step": 3105, "epoch": 0.3119035660472125, "loss": 0.0259, "grad_norm": 12.877971649169922, "learning_rate": 5.93939393939394e-07, "num_tokens": 5819322.0, "completions/mean_length": 60.875, "completions/min_length": 55.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.8762128949165344, "rewards/meter/std": 0.3151567280292511, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.984167218208313, "rewards/repeat_soft/std": 0.01779637299478054, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.7645875215530396, "rewards/total_composite/std": 0.14070084691047668, "reward": 0.7645875215530396, "reward_std": 0.14070084691047668, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08938949555158615, "sampling/sampling_logp_difference/max": 1.117671012878418, "sampling/importance_sampling_ratio/min": 0.32704058289527893, "sampling/importance_sampling_ratio/mean": 1.007515788078308, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5111999437212944, "clip_ratio/low_mean": 0.010080644860863686, "clip_ratio/low_min": 0.010080644860863686, "clip_ratio/high_mean": 0.091474998742342, "clip_ratio/high_max": 0.091474998742342, "clip_ratio/region_mean": 0.10155564360320568, "reward_total_mean": 0.7645875215530396, "reward_meter_mean": 0.8762128949165344, "reward_meter_std": 0.3151567280292511, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.984167218208313, "reward_repeat_soft_std": 0.01779637299478054, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.7645875215530396, "reward_total_composite_std": 0.14070084691047668} {"timestamp_utc": "2026-04-13T06:08:16Z", "mode": "train", "global_step": 3106, "epoch": 0.31200401808136613, "loss": 0.0134, "grad_norm": 6.2390336990356445, "learning_rate": 5.90909090909091e-07, "num_tokens": 5821786.0, "completions/mean_length": 129.0, "completions/min_length": 118.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 129.0, "completions/min_terminated_length": 118.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.6558979749679565, "rewards/meter/std": 0.2682810425758362, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9227975606918335, "rewards/repeat_soft/std": 0.02959430031478405, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.6506838798522949, "rewards/total_composite/std": 0.13105323910713196, "reward": 0.6506838798522949, "reward_std": 0.13105323910713196, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09508421272039413, "sampling/sampling_logp_difference/max": 2.9087014198303223, "sampling/importance_sampling_ratio/min": 0.05454651638865471, "sampling/importance_sampling_ratio/mean": 0.9974325895309448, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40709632635116577, "clip_ratio/low_mean": 0.01994946552440524, "clip_ratio/low_min": 0.01994946552440524, "clip_ratio/high_mean": 0.06697289552539587, "clip_ratio/high_max": 0.06697289552539587, "clip_ratio/region_mean": 0.08692236104980111, "reward_total_mean": 0.6506838798522949, "reward_meter_mean": 0.6558979749679565, "reward_meter_std": 0.2682810425758362, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9227975606918335, "reward_repeat_soft_std": 0.02959430031478405, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.6506838798522949, "reward_total_composite_std": 0.13105323910713196} {"timestamp_utc": "2026-04-13T06:08:22Z", "mode": "train", "global_step": 3107, "epoch": 0.31210447011551984, "loss": 0.0104, "grad_norm": 8.878045082092285, "learning_rate": 5.878787878787879e-07, "num_tokens": 5823471.0, "completions/mean_length": 51.625, "completions/min_length": 49.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.625, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.9762858152389526, "rewards/meter/std": 0.02862318977713585, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9962759017944336, "rewards/repeat_soft/std": 0.007037608418613672, "rewards/judge_quality/mean": 0.6449999809265137, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.8824561834335327, "rewards/total_composite/std": 0.04176553711295128, "reward": 0.8824561834335327, "reward_std": 0.041765518486499786, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10542553663253784, "sampling/sampling_logp_difference/max": 2.7900443077087402, "sampling/importance_sampling_ratio/min": 0.06141849234700203, "sampling/importance_sampling_ratio/mean": 0.9824560284614563, "sampling/importance_sampling_ratio/max": 1.923088550567627, "entropy": 0.4109141454100609, "clip_ratio/low_mean": 0.027506002690643072, "clip_ratio/low_min": 0.027506002690643072, "clip_ratio/high_mean": 0.07374818716198206, "clip_ratio/high_max": 0.07374818716198206, "clip_ratio/region_mean": 0.10125418985262513, "reward_total_mean": 0.8824561834335327, "reward_meter_mean": 0.9762858152389526, "reward_meter_std": 0.02862318977713585, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9962759017944336, "reward_repeat_soft_std": 0.007037608418613672, "reward_judge_quality_mean": 0.6449999809265137, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.8824561834335327, "reward_total_composite_std": 0.04176553711295128} {"timestamp_utc": "2026-04-13T06:08:28Z", "mode": "train", "global_step": 3108, "epoch": 0.31220492214967355, "loss": 0.0352, "grad_norm": 11.806609153747559, "learning_rate": 5.848484848484849e-07, "num_tokens": 5824977.0, "completions/mean_length": 30.25, "completions/min_length": 29.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.25, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9867740869522095, "rewards/meter/std": 0.01110746804624796, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9545454382896423, "rewards/repeat_soft/std": 0.014728941954672337, "rewards/judge_quality/mean": 0.3962499797344208, "rewards/judge_quality/std": 0.09085899591445923, "rewards/total_composite/mean": 0.8083779215812683, "rewards/total_composite/std": 0.029304316267371178, "reward": 0.8083779215812683, "reward_std": 0.029304319992661476, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10352708399295807, "sampling/sampling_logp_difference/max": 1.7449836730957031, "sampling/importance_sampling_ratio/min": 0.17464783787727356, "sampling/importance_sampling_ratio/mean": 1.0097616910934448, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4642808139324188, "clip_ratio/low_mean": 0.019791666883975267, "clip_ratio/low_min": 0.019791666883975267, "clip_ratio/high_mean": 0.08387424005195498, "clip_ratio/high_max": 0.08387424005195498, "clip_ratio/region_mean": 0.10366590693593025, "reward_total_mean": 0.8083779215812683, "reward_meter_mean": 0.9867740869522095, "reward_meter_std": 0.01110746804624796, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9545454382896423, "reward_repeat_soft_std": 0.014728941954672337, "reward_judge_quality_mean": 0.3962499797344208, "reward_judge_quality_std": 0.09085899591445923, "reward_total_composite_mean": 0.8083779215812683, "reward_total_composite_std": 0.029304316267371178} {"timestamp_utc": "2026-04-13T06:08:35Z", "mode": "train", "global_step": 3109, "epoch": 0.3123053741838272, "loss": 0.0445, "grad_norm": 7.754781723022461, "learning_rate": 5.818181818181819e-07, "num_tokens": 5827109.0, "completions/mean_length": 83.5, "completions/min_length": 72.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.5, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.9832293391227722, "rewards/meter/std": 0.005043665412813425, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9506518840789795, "rewards/repeat_soft/std": 0.027794724330306053, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.8885183930397034, "rewards/total_composite/std": 0.08175535500049591, "reward": 0.8885183930397034, "reward_std": 0.08175535500049591, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07202117145061493, "sampling/sampling_logp_difference/max": 1.6245172023773193, "sampling/importance_sampling_ratio/min": 0.41234180331230164, "sampling/importance_sampling_ratio/mean": 1.0125281810760498, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.33897850662469864, "clip_ratio/low_mean": 0.031878076028078794, "clip_ratio/low_min": 0.031878076028078794, "clip_ratio/high_mean": 0.037042805925011635, "clip_ratio/high_max": 0.037042805925011635, "clip_ratio/region_mean": 0.06892088195309043, "reward_total_mean": 0.8885183930397034, "reward_meter_mean": 0.9832293391227722, "reward_meter_std": 0.005043665412813425, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9506518840789795, "reward_repeat_soft_std": 0.027794724330306053, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.8885183930397034, "reward_total_composite_std": 0.08175535500049591} {"timestamp_utc": "2026-04-13T06:08:42Z", "mode": "train", "global_step": 3110, "epoch": 0.3124058262179809, "loss": 0.0175, "grad_norm": 10.452248573303223, "learning_rate": 5.787878787878789e-07, "num_tokens": 5828839.0, "completions/mean_length": 55.25, "completions/min_length": 52.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.25, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9920142889022827, "rewards/meter/std": 0.0012652826262637973, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9890224933624268, "rewards/repeat_soft/std": 0.006785218138247728, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.8610587120056152, "rewards/total_composite/std": 0.06841269880533218, "reward": 0.8610587120056152, "reward_std": 0.06841269135475159, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07584834843873978, "sampling/sampling_logp_difference/max": 1.0819268226623535, "sampling/importance_sampling_ratio/min": 0.3389417827129364, "sampling/importance_sampling_ratio/mean": 1.0236502885818481, "sampling/importance_sampling_ratio/max": 1.9440312385559082, "entropy": 0.41302576288580894, "clip_ratio/low_mean": 0.08815762866288424, "clip_ratio/low_min": 0.08815762866288424, "clip_ratio/high_mean": 0.015787337440997362, "clip_ratio/high_max": 0.015787337440997362, "clip_ratio/region_mean": 0.1039449661038816, "reward_total_mean": 0.8610587120056152, "reward_meter_mean": 0.9920142889022827, "reward_meter_std": 0.0012652826262637973, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9890224933624268, "reward_repeat_soft_std": 0.006785218138247728, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.8610587120056152, "reward_total_composite_std": 0.06841269880533218} {"timestamp_utc": "2026-04-13T06:08:49Z", "mode": "train", "global_step": 3111, "epoch": 0.3125062782521346, "loss": 0.0335, "grad_norm": 11.304492950439453, "learning_rate": 5.757575757575758e-07, "num_tokens": 5830751.0, "completions/mean_length": 62.0, "completions/min_length": 56.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.8742128610610962, "rewards/meter/std": 0.2712404131889343, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9658246040344238, "rewards/repeat_soft/std": 0.05984355881810188, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.8046032190322876, "rewards/total_composite/std": 0.15611693263053894, "reward": 0.8046032190322876, "reward_std": 0.15611694753170013, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10321196913719177, "sampling/sampling_logp_difference/max": 3.1803345680236816, "sampling/importance_sampling_ratio/min": 0.04157174378633499, "sampling/importance_sampling_ratio/mean": 1.0135468244552612, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.41895730048418045, "clip_ratio/low_mean": 0.05003450810909271, "clip_ratio/low_min": 0.05003450810909271, "clip_ratio/high_mean": 0.04132395493797958, "clip_ratio/high_max": 0.04132395493797958, "clip_ratio/region_mean": 0.09135846304707229, "reward_total_mean": 0.8046032190322876, "reward_meter_mean": 0.8742128610610962, "reward_meter_std": 0.2712404131889343, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9658246040344238, "reward_repeat_soft_std": 0.05984355881810188, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.8046032190322876, "reward_total_composite_std": 0.15611693263053894} {"timestamp_utc": "2026-04-13T06:09:01Z", "mode": "train", "global_step": 3112, "epoch": 0.31260673028628827, "loss": -0.1805, "grad_norm": 1.9560374021530151, "learning_rate": 5.727272727272728e-07, "num_tokens": 5832661.0, "completions/mean_length": 140.75, "completions/min_length": 72.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 87.71428680419922, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.861810564994812, "rewards/meter/std": 0.34826982021331787, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9856328964233398, "rewards/repeat_soft/std": 0.011042882688343525, "rewards/judge_quality/mean": 0.6237499713897705, "rewards/judge_quality/std": 0.33907175064086914, "rewards/total_composite/mean": 0.7903780341148376, "rewards/total_composite/std": 0.3281303644180298, "reward": 0.7903780341148376, "reward_std": 0.3281303346157074, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.085739865899086, "sampling/sampling_logp_difference/max": 1.48272705078125, "sampling/importance_sampling_ratio/min": 0.22701776027679443, "sampling/importance_sampling_ratio/mean": 1.0022411346435547, "sampling/importance_sampling_ratio/max": 1.9058383703231812, "entropy": 0.34645354375243187, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06377363251522183, "clip_ratio/high_max": 0.06377363251522183, "clip_ratio/region_mean": 0.06377363251522183, "reward_total_mean": 0.7903780341148376, "reward_meter_mean": 0.861810564994812, "reward_meter_std": 0.34826982021331787, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9856328964233398, "reward_repeat_soft_std": 0.011042882688343525, "reward_judge_quality_mean": 0.6237499713897705, "reward_judge_quality_std": 0.33907175064086914, "reward_total_composite_mean": 0.7903780341148376, "reward_total_composite_std": 0.3281303644180298} {"timestamp_utc": "2026-04-13T06:09:12Z", "mode": "train", "global_step": 3113, "epoch": 0.312707182320442, "loss": -0.2134, "grad_norm": 2.0478169918060303, "learning_rate": 5.696969696969698e-07, "num_tokens": 5834919.0, "completions/mean_length": 171.25, "completions/min_length": 102.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 122.5714340209961, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.9726045727729797, "rewards/meter/std": 0.046304430812597275, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9127157926559448, "rewards/repeat_soft/std": 0.040836963802576065, "rewards/judge_quality/mean": 0.3187499940395355, "rewards/judge_quality/std": 0.14961259067058563, "rewards/total_composite/mean": 0.6889399290084839, "rewards/total_composite/std": 0.28040972352027893, "reward": 0.6889399290084839, "reward_std": 0.28040972352027893, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08158749341964722, "sampling/sampling_logp_difference/max": 1.4537218809127808, "sampling/importance_sampling_ratio/min": 0.23369885981082916, "sampling/importance_sampling_ratio/mean": 1.0091572999954224, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3460669182240963, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06493690516799688, "clip_ratio/high_max": 0.06493690516799688, "clip_ratio/region_mean": 0.06493690516799688, "reward_total_mean": 0.6889399290084839, "reward_meter_mean": 0.9726045727729797, "reward_meter_std": 0.046304430812597275, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9127157926559448, "reward_repeat_soft_std": 0.040836963802576065, "reward_judge_quality_mean": 0.3187499940395355, "reward_judge_quality_std": 0.14961259067058563, "reward_total_composite_mean": 0.6889399290084839, "reward_total_composite_std": 0.28040972352027893} {"timestamp_utc": "2026-04-13T06:09:18Z", "mode": "train", "global_step": 3114, "epoch": 0.3128076343545957, "loss": 0.0467, "grad_norm": 12.849146842956543, "learning_rate": 5.666666666666667e-07, "num_tokens": 5836230.0, "completions/mean_length": 27.875, "completions/min_length": 23.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.875, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9321378469467163, "rewards/meter/std": 0.16829681396484375, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9615821838378906, "rewards/repeat_soft/std": 0.002596014179289341, "rewards/judge_quality/mean": 0.3799999952316284, "rewards/judge_quality/std": 0.11501552164554596, "rewards/total_composite/mean": 0.7796202301979065, "rewards/total_composite/std": 0.07450763136148453, "reward": 0.7796202301979065, "reward_std": 0.07450763136148453, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0822710245847702, "sampling/sampling_logp_difference/max": 0.9516067504882812, "sampling/importance_sampling_ratio/min": 0.38612014055252075, "sampling/importance_sampling_ratio/mean": 1.0176136493682861, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5384289957582951, "clip_ratio/low_mean": 0.012568137608468533, "clip_ratio/low_min": 0.012568137608468533, "clip_ratio/high_mean": 0.054188772570341825, "clip_ratio/high_max": 0.054188772570341825, "clip_ratio/region_mean": 0.06675691017881036, "reward_total_mean": 0.7796202301979065, "reward_meter_mean": 0.9321378469467163, "reward_meter_std": 0.16829681396484375, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9615821838378906, "reward_repeat_soft_std": 0.002596014179289341, "reward_judge_quality_mean": 0.3799999952316284, "reward_judge_quality_std": 0.11501552164554596, "reward_total_composite_mean": 0.7796202301979065, "reward_total_composite_std": 0.07450763136148453} {"timestamp_utc": "2026-04-13T06:09:25Z", "mode": "train", "global_step": 3115, "epoch": 0.3129080863887494, "loss": 0.0361, "grad_norm": 20.549509048461914, "learning_rate": 5.636363636363638e-07, "num_tokens": 5837712.0, "completions/mean_length": 30.25, "completions/min_length": 27.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.25, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9929935336112976, "rewards/meter/std": 0.002612440614029765, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9424828290939331, "rewards/repeat_soft/std": 0.019883159548044205, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.14320313930511475, "rewards/total_composite/mean": 0.7878453731536865, "rewards/total_composite/std": 0.04100780934095383, "reward": 0.7878453731536865, "reward_std": 0.041007813066244125, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07093393057584763, "sampling/sampling_logp_difference/max": 0.8880754709243774, "sampling/importance_sampling_ratio/min": 0.41144683957099915, "sampling/importance_sampling_ratio/mean": 1.001691222190857, "sampling/importance_sampling_ratio/max": 1.5691723823547363, "entropy": 0.3773055635392666, "clip_ratio/low_mean": 0.012408568523824215, "clip_ratio/low_min": 0.012408568523824215, "clip_ratio/high_mean": 0.025059737265110016, "clip_ratio/high_max": 0.025059737265110016, "clip_ratio/region_mean": 0.03746830578893423, "reward_total_mean": 0.7878453731536865, "reward_meter_mean": 0.9929935336112976, "reward_meter_std": 0.002612440614029765, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9424828290939331, "reward_repeat_soft_std": 0.019883159548044205, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.14320313930511475, "reward_total_composite_mean": 0.7878453731536865, "reward_total_composite_std": 0.04100780934095383} {"timestamp_utc": "2026-04-13T06:09:32Z", "mode": "train", "global_step": 3116, "epoch": 0.31300853842290305, "loss": 0.0421, "grad_norm": 8.717110633850098, "learning_rate": 5.606060606060607e-07, "num_tokens": 5839596.0, "completions/mean_length": 80.5, "completions/min_length": 72.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.5, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9793050289154053, "rewards/meter/std": 0.011908333748579025, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9514926075935364, "rewards/repeat_soft/std": 0.030539652332663536, "rewards/judge_quality/mean": 0.4987499713897705, "rewards/judge_quality/std": 0.13695022463798523, "rewards/total_composite/mean": 0.8354614973068237, "rewards/total_composite/std": 0.04461805522441864, "reward": 0.8354614973068237, "reward_std": 0.04461806267499924, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10346166789531708, "sampling/sampling_logp_difference/max": 1.8094592094421387, "sampling/importance_sampling_ratio/min": 0.16374266147613525, "sampling/importance_sampling_ratio/mean": 0.9977829456329346, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43601537868380547, "clip_ratio/low_mean": 0.06402429286390543, "clip_ratio/low_min": 0.06402429286390543, "clip_ratio/high_mean": 0.023891063407063484, "clip_ratio/high_max": 0.023891063407063484, "clip_ratio/region_mean": 0.08791535627096891, "reward_total_mean": 0.8354614973068237, "reward_meter_mean": 0.9793050289154053, "reward_meter_std": 0.011908333748579025, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9514926075935364, "reward_repeat_soft_std": 0.030539652332663536, "reward_judge_quality_mean": 0.4987499713897705, "reward_judge_quality_std": 0.13695022463798523, "reward_total_composite_mean": 0.8354614973068237, "reward_total_composite_std": 0.04461805522441864} {"timestamp_utc": "2026-04-13T06:09:38Z", "mode": "train", "global_step": 3117, "epoch": 0.31310899045705676, "loss": 0.012, "grad_norm": 7.417283535003662, "learning_rate": 5.575757575757576e-07, "num_tokens": 5841345.0, "completions/mean_length": 59.625, "completions/min_length": 57.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.625, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9871981143951416, "rewards/meter/std": 0.004603895358741283, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9705605506896973, "rewards/repeat_soft/std": 0.04192313551902771, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.8371702432632446, "rewards/total_composite/std": 0.054452084004879, "reward": 0.8371702432632446, "reward_std": 0.054452065378427505, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07486163079738617, "sampling/sampling_logp_difference/max": 1.2935127019882202, "sampling/importance_sampling_ratio/min": 0.274305522441864, "sampling/importance_sampling_ratio/mean": 1.0075771808624268, "sampling/importance_sampling_ratio/max": 1.7924562692642212, "entropy": 0.4394451081752777, "clip_ratio/low_mean": 0.06091459677554667, "clip_ratio/low_min": 0.06091459677554667, "clip_ratio/high_mean": 0.012500000186264515, "clip_ratio/high_max": 0.012500000186264515, "clip_ratio/region_mean": 0.07341459696181118, "reward_total_mean": 0.8371702432632446, "reward_meter_mean": 0.9871981143951416, "reward_meter_std": 0.004603895358741283, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9705605506896973, "reward_repeat_soft_std": 0.04192313551902771, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.8371702432632446, "reward_total_composite_std": 0.054452084004879} {"timestamp_utc": "2026-04-13T06:09:50Z", "mode": "train", "global_step": 3118, "epoch": 0.31320944249121047, "loss": -0.2337, "grad_norm": 1.0030536651611328, "learning_rate": 5.545454545454547e-07, "num_tokens": 5843850.0, "completions/mean_length": 193.125, "completions/min_length": 138.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 147.57144165039062, "completions/min_terminated_length": 138.0, "completions/max_terminated_length": 163.0, "rewards/meter/mean": 0.9747388362884521, "rewards/meter/std": 0.004631054122000933, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.17251639068126678, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7248746156692505, "rewards/repeat_soft/std": 0.09353797137737274, "rewards/judge_quality/mean": 0.2212499976158142, "rewards/judge_quality/std": 0.10842212289571762, "rewards/total_composite/mean": 0.6327506303787231, "rewards/total_composite/std": 0.2564191222190857, "reward": 0.6327506303787231, "reward_std": 0.2564191222190857, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.035678062587976456, "sampling/sampling_logp_difference/max": 0.9595696926116943, "sampling/importance_sampling_ratio/min": 0.38305771350860596, "sampling/importance_sampling_ratio/mean": 1.0081937313079834, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17130953446030617, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.024813265539705753, "clip_ratio/high_max": 0.024813265539705753, "clip_ratio/region_mean": 0.024813265539705753, "reward_total_mean": 0.6327506303787231, "reward_meter_mean": 0.9747388362884521, "reward_meter_std": 0.004631054122000933, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.17251639068126678, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7248746156692505, "reward_repeat_soft_std": 0.09353797137737274, "reward_judge_quality_mean": 0.2212499976158142, "reward_judge_quality_std": 0.10842212289571762, "reward_total_composite_mean": 0.6327506303787231, "reward_total_composite_std": 0.2564191222190857} {"timestamp_utc": "2026-04-13T06:09:56Z", "mode": "train", "global_step": 3119, "epoch": 0.3133098945253641, "loss": 0.059, "grad_norm": 10.361119270324707, "learning_rate": 5.515151515151516e-07, "num_tokens": 5845643.0, "completions/mean_length": 44.125, "completions/min_length": 41.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.125, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9598516225814819, "rewards/meter/std": 0.009078252129256725, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9492527842521667, "rewards/repeat_soft/std": 0.028999656438827515, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.8403584957122803, "rewards/total_composite/std": 0.0676536038517952, "reward": 0.8403584957122803, "reward_std": 0.067653588950634, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08193644136190414, "sampling/sampling_logp_difference/max": 1.1070752143859863, "sampling/importance_sampling_ratio/min": 0.3305242657661438, "sampling/importance_sampling_ratio/mean": 1.0048518180847168, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36008502170443535, "clip_ratio/low_mean": 0.053033245261758566, "clip_ratio/low_min": 0.053033245261758566, "clip_ratio/high_mean": 0.021123693324625492, "clip_ratio/high_max": 0.021123693324625492, "clip_ratio/region_mean": 0.07415693858638406, "reward_total_mean": 0.8403584957122803, "reward_meter_mean": 0.9598516225814819, "reward_meter_std": 0.009078252129256725, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9492527842521667, "reward_repeat_soft_std": 0.028999656438827515, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.8403584957122803, "reward_total_composite_std": 0.0676536038517952} {"timestamp_utc": "2026-04-13T06:10:04Z", "mode": "train", "global_step": 3120, "epoch": 0.3134103465595178, "loss": 0.0152, "grad_norm": 5.116827964782715, "learning_rate": 5.484848484848485e-07, "num_tokens": 5848534.0, "completions/mean_length": 180.375, "completions/min_length": 176.0, "completions/max_length": 189.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 180.375, "completions/min_terminated_length": 176.0, "completions/max_terminated_length": 189.0, "rewards/meter/mean": 0.990037202835083, "rewards/meter/std": 0.006464237347245216, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8439759612083435, "rewards/repeat_soft/std": 0.05568486452102661, "rewards/judge_quality/mean": 0.23000000417232513, "rewards/judge_quality/std": 0.12224097549915314, "rewards/total_composite/mean": 0.7489143013954163, "rewards/total_composite/std": 0.03485932573676109, "reward": 0.7489143013954163, "reward_std": 0.034859322011470795, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07171020656824112, "sampling/sampling_logp_difference/max": 1.7901403903961182, "sampling/importance_sampling_ratio/min": 0.1669367402791977, "sampling/importance_sampling_ratio/mean": 1.0064085721969604, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3806404061615467, "clip_ratio/low_mean": 0.04319977783598006, "clip_ratio/low_min": 0.04319977783598006, "clip_ratio/high_mean": 0.024406705982983112, "clip_ratio/high_max": 0.024406705982983112, "clip_ratio/region_mean": 0.06760648381896317, "reward_total_mean": 0.7489143013954163, "reward_meter_mean": 0.990037202835083, "reward_meter_std": 0.006464237347245216, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8439759612083435, "reward_repeat_soft_std": 0.05568486452102661, "reward_judge_quality_mean": 0.23000000417232513, "reward_judge_quality_std": 0.12224097549915314, "reward_total_composite_mean": 0.7489143013954163, "reward_total_composite_std": 0.03485932573676109} {"timestamp_utc": "2026-04-13T06:10:12Z", "mode": "train", "global_step": 3121, "epoch": 0.31351079859367154, "loss": 0.0323, "grad_norm": 7.048356056213379, "learning_rate": 5.454545454545455e-07, "num_tokens": 5851133.0, "completions/mean_length": 147.875, "completions/min_length": 139.0, "completions/max_length": 157.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 147.875, "completions/min_terminated_length": 139.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.9734389185905457, "rewards/meter/std": 0.055048756301403046, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9418433308601379, "rewards/repeat_soft/std": 0.036869872361421585, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.8018568754196167, "rewards/total_composite/std": 0.02960047498345375, "reward": 0.8018568754196167, "reward_std": 0.02960047498345375, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09776268154382706, "sampling/sampling_logp_difference/max": 1.726090431213379, "sampling/importance_sampling_ratio/min": 0.17797887325286865, "sampling/importance_sampling_ratio/mean": 1.0066324472427368, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4894905537366867, "clip_ratio/low_mean": 0.02416286151856184, "clip_ratio/low_min": 0.02416286151856184, "clip_ratio/high_mean": 0.07051508082076907, "clip_ratio/high_max": 0.07051508082076907, "clip_ratio/region_mean": 0.09467794233933091, "reward_total_mean": 0.8018568754196167, "reward_meter_mean": 0.9734389185905457, "reward_meter_std": 0.055048756301403046, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9418433308601379, "reward_repeat_soft_std": 0.036869872361421585, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.8018568754196167, "reward_total_composite_std": 0.02960047498345375} {"timestamp_utc": "2026-04-13T06:10:18Z", "mode": "train", "global_step": 3122, "epoch": 0.3136112506278252, "loss": -0.0162, "grad_norm": 10.232865333557129, "learning_rate": 5.424242424242425e-07, "num_tokens": 5852584.0, "completions/mean_length": 31.375, "completions/min_length": 29.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.375, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9782928228378296, "rewards/meter/std": 0.015957670286297798, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9461931586265564, "rewards/repeat_soft/std": 0.025219468399882317, "rewards/judge_quality/mean": 0.44624999165534973, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8187260627746582, "rewards/total_composite/std": 0.007582070305943489, "reward": 0.8187260627746582, "reward_std": 0.0075820679776370525, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09504683315753937, "sampling/sampling_logp_difference/max": 0.7951661944389343, "sampling/importance_sampling_ratio/min": 0.45150619745254517, "sampling/importance_sampling_ratio/mean": 1.0284876823425293, "sampling/importance_sampling_ratio/max": 1.935731053352356, "entropy": 0.498825129121542, "clip_ratio/low_mean": 0.030172414146363735, "clip_ratio/low_min": 0.030172414146363735, "clip_ratio/high_mean": 0.08541666809469461, "clip_ratio/high_max": 0.08541666809469461, "clip_ratio/region_mean": 0.11558908224105835, "reward_total_mean": 0.8187260627746582, "reward_meter_mean": 0.9782928228378296, "reward_meter_std": 0.015957670286297798, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9461931586265564, "reward_repeat_soft_std": 0.025219468399882317, "reward_judge_quality_mean": 0.44624999165534973, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8187260627746582, "reward_total_composite_std": 0.007582070305943489} {"timestamp_utc": "2026-04-13T06:10:26Z", "mode": "train", "global_step": 3123, "epoch": 0.3137117026619789, "loss": 0.0064, "grad_norm": 5.090183734893799, "learning_rate": 5.393939393939395e-07, "num_tokens": 5855045.0, "completions/mean_length": 123.625, "completions/min_length": 112.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 123.625, "completions/min_terminated_length": 112.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.9907081723213196, "rewards/meter/std": 0.0033634433057159185, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9312554001808167, "rewards/repeat_soft/std": 0.03241767734289169, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8149442076683044, "rewards/total_composite/std": 0.0030315981712192297, "reward": 0.8149442076683044, "reward_std": 0.003031595144420862, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07428034394979477, "sampling/sampling_logp_difference/max": 1.612687587738037, "sampling/importance_sampling_ratio/min": 0.19935111701488495, "sampling/importance_sampling_ratio/mean": 1.000434160232544, "sampling/importance_sampling_ratio/max": 1.9899288415908813, "entropy": 0.33713531866669655, "clip_ratio/low_mean": 0.008348214207217097, "clip_ratio/low_min": 0.008348214207217097, "clip_ratio/high_mean": 0.05746041378006339, "clip_ratio/high_max": 0.05746041378006339, "clip_ratio/region_mean": 0.06580862798728049, "reward_total_mean": 0.8149442076683044, "reward_meter_mean": 0.9907081723213196, "reward_meter_std": 0.0033634433057159185, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9312554001808167, "reward_repeat_soft_std": 0.03241767734289169, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8149442076683044, "reward_total_composite_std": 0.0030315981712192297} {"timestamp_utc": "2026-04-13T06:10:32Z", "mode": "train", "global_step": 3124, "epoch": 0.3138121546961326, "loss": 0.0054, "grad_norm": 7.247985363006592, "learning_rate": 5.363636363636364e-07, "num_tokens": 5856901.0, "completions/mean_length": 67.0, "completions/min_length": 65.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9939215183258057, "rewards/meter/std": 0.002999773481860757, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9649595022201538, "rewards/repeat_soft/std": 0.03424415737390518, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.1865811049938202, "rewards/total_composite/mean": 0.8531356453895569, "rewards/total_composite/std": 0.053969837725162506, "reward": 0.8531356453895569, "reward_std": 0.05396983027458191, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08983299881219864, "sampling/sampling_logp_difference/max": 2.1155648231506348, "sampling/importance_sampling_ratio/min": 0.12056516855955124, "sampling/importance_sampling_ratio/mean": 1.0073039531707764, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4668492488563061, "clip_ratio/low_mean": 0.06183908227831125, "clip_ratio/low_min": 0.06183908227831125, "clip_ratio/high_mean": 0.023374865297228098, "clip_ratio/high_max": 0.023374865297228098, "clip_ratio/region_mean": 0.08521394757553935, "reward_total_mean": 0.8531356453895569, "reward_meter_mean": 0.9939215183258057, "reward_meter_std": 0.002999773481860757, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9649595022201538, "reward_repeat_soft_std": 0.03424415737390518, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.1865811049938202, "reward_total_composite_mean": 0.8531356453895569, "reward_total_composite_std": 0.053969837725162506} {"timestamp_utc": "2026-04-13T06:10:38Z", "mode": "train", "global_step": 3125, "epoch": 0.3139126067302863, "loss": 0.0237, "grad_norm": 13.775800704956055, "learning_rate": 5.333333333333335e-07, "num_tokens": 5858462.0, "completions/mean_length": 49.125, "completions/min_length": 46.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.125, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9207918047904968, "rewards/meter/std": 0.14191533625125885, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9627680778503418, "rewards/repeat_soft/std": 0.037830650806427, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.8053830862045288, "rewards/total_composite/std": 0.09051623940467834, "reward": 0.8053830862045288, "reward_std": 0.09051623195409775, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09661808609962463, "sampling/sampling_logp_difference/max": 1.6052435636520386, "sampling/importance_sampling_ratio/min": 0.20084063708782196, "sampling/importance_sampling_ratio/mean": 0.9911250472068787, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4955285005271435, "clip_ratio/low_mean": 0.025109540671110153, "clip_ratio/low_min": 0.025109540671110153, "clip_ratio/high_mean": 0.07552803819999099, "clip_ratio/high_max": 0.07552803819999099, "clip_ratio/region_mean": 0.10063757887110114, "reward_total_mean": 0.8053830862045288, "reward_meter_mean": 0.9207918047904968, "reward_meter_std": 0.14191533625125885, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9627680778503418, "reward_repeat_soft_std": 0.037830650806427, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.8053830862045288, "reward_total_composite_std": 0.09051623940467834} {"timestamp_utc": "2026-04-13T06:10:46Z", "mode": "train", "global_step": 3126, "epoch": 0.31401305876443997, "loss": 0.002, "grad_norm": 8.103065490722656, "learning_rate": 5.303030303030304e-07, "num_tokens": 5860267.0, "completions/mean_length": 67.625, "completions/min_length": 61.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.625, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9932055473327637, "rewards/meter/std": 0.0018408889882266521, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9778234958648682, "rewards/repeat_soft/std": 0.020811256021261215, "rewards/judge_quality/mean": 0.6012499928474426, "rewards/judge_quality/std": 0.2423950731754303, "rewards/total_composite/mean": 0.8750998377799988, "rewards/total_composite/std": 0.07239512354135513, "reward": 0.8750998377799988, "reward_std": 0.07239512354135513, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09137352555990219, "sampling/sampling_logp_difference/max": 1.7647509574890137, "sampling/importance_sampling_ratio/min": 0.22527344524860382, "sampling/importance_sampling_ratio/mean": 1.0144022703170776, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4029795303940773, "clip_ratio/low_mean": 0.03585443925112486, "clip_ratio/low_min": 0.03585443925112486, "clip_ratio/high_mean": 0.030637255404144526, "clip_ratio/high_max": 0.030637255404144526, "clip_ratio/region_mean": 0.06649169465526938, "reward_total_mean": 0.8750998377799988, "reward_meter_mean": 0.9932055473327637, "reward_meter_std": 0.0018408889882266521, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9778234958648682, "reward_repeat_soft_std": 0.020811256021261215, "reward_judge_quality_mean": 0.6012499928474426, "reward_judge_quality_std": 0.2423950731754303, "reward_total_composite_mean": 0.8750998377799988, "reward_total_composite_std": 0.07239512354135513} {"timestamp_utc": "2026-04-13T06:10:54Z", "mode": "train", "global_step": 3127, "epoch": 0.3141135107985937, "loss": -0.0057, "grad_norm": 4.923807144165039, "learning_rate": 5.272727272727273e-07, "num_tokens": 5863224.0, "completions/mean_length": 163.625, "completions/min_length": 151.0, "completions/max_length": 180.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 163.625, "completions/min_terminated_length": 151.0, "completions/max_terminated_length": 180.0, "rewards/meter/mean": 0.9869558811187744, "rewards/meter/std": 0.013414217159152031, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7342267632484436, "rewards/repeat_soft/std": 0.11967334151268005, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.7834278345108032, "rewards/total_composite/std": 0.022894281893968582, "reward": 0.7834278345108032, "reward_std": 0.022894278168678284, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07458927482366562, "sampling/sampling_logp_difference/max": 2.5764248371124268, "sampling/importance_sampling_ratio/min": 0.07604539394378662, "sampling/importance_sampling_ratio/mean": 0.9953667521476746, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3445831276476383, "clip_ratio/low_mean": 0.011867088731378317, "clip_ratio/low_min": 0.011867088731378317, "clip_ratio/high_mean": 0.06402223091572523, "clip_ratio/high_max": 0.06402223091572523, "clip_ratio/region_mean": 0.07588931964710355, "reward_total_mean": 0.7834278345108032, "reward_meter_mean": 0.9869558811187744, "reward_meter_std": 0.013414217159152031, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7342267632484436, "reward_repeat_soft_std": 0.11967334151268005, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.7834278345108032, "reward_total_composite_std": 0.022894281893968582} {"timestamp_utc": "2026-04-13T06:11:00Z", "mode": "train", "global_step": 3128, "epoch": 0.3142139628327474, "loss": 0.0037, "grad_norm": 9.084616661071777, "learning_rate": 5.242424242424243e-07, "num_tokens": 5864942.0, "completions/mean_length": 56.75, "completions/min_length": 54.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.75, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9877734184265137, "rewards/meter/std": 0.00401577353477478, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9550327062606812, "rewards/repeat_soft/std": 0.037941861897706985, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.834751307964325, "rewards/total_composite/std": 0.052706632763147354, "reward": 0.834751307964325, "reward_std": 0.05270661041140556, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0742814764380455, "sampling/sampling_logp_difference/max": 1.2292990684509277, "sampling/importance_sampling_ratio/min": 0.29249751567840576, "sampling/importance_sampling_ratio/mean": 1.0107486248016357, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37379027903079987, "clip_ratio/low_mean": 0.05818359926342964, "clip_ratio/low_min": 0.05818359926342964, "clip_ratio/high_mean": 0.008474576286971569, "clip_ratio/high_max": 0.008474576286971569, "clip_ratio/region_mean": 0.06665817555040121, "reward_total_mean": 0.834751307964325, "reward_meter_mean": 0.9877734184265137, "reward_meter_std": 0.00401577353477478, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9550327062606812, "reward_repeat_soft_std": 0.037941861897706985, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.834751307964325, "reward_total_composite_std": 0.052706632763147354} {"timestamp_utc": "2026-04-13T06:11:07Z", "mode": "train", "global_step": 3129, "epoch": 0.31431441486690104, "loss": 0.0108, "grad_norm": 7.628948211669922, "learning_rate": 5.212121212121213e-07, "num_tokens": 5866759.0, "completions/mean_length": 62.125, "completions/min_length": 57.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.125, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9596515893936157, "rewards/meter/std": 0.08130810409784317, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9506441950798035, "rewards/repeat_soft/std": 0.04608358070254326, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8085326552391052, "rewards/total_composite/std": 0.0343761220574379, "reward": 0.8085326552391052, "reward_std": 0.0343761220574379, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08026956021785736, "sampling/sampling_logp_difference/max": 1.3530449867248535, "sampling/importance_sampling_ratio/min": 0.25845205783843994, "sampling/importance_sampling_ratio/mean": 0.9956100583076477, "sampling/importance_sampling_ratio/max": 1.8631130456924438, "entropy": 0.363037820905447, "clip_ratio/low_mean": 0.02016128972172737, "clip_ratio/low_min": 0.02016128972172737, "clip_ratio/high_mean": 0.08304394781589508, "clip_ratio/high_max": 0.08304394781589508, "clip_ratio/region_mean": 0.10320523753762245, "reward_total_mean": 0.8085326552391052, "reward_meter_mean": 0.9596515893936157, "reward_meter_std": 0.08130810409784317, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9506441950798035, "reward_repeat_soft_std": 0.04608358070254326, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8085326552391052, "reward_total_composite_std": 0.0343761220574379} {"timestamp_utc": "2026-04-13T06:11:13Z", "mode": "train", "global_step": 3130, "epoch": 0.31441486690105475, "loss": -0.0206, "grad_norm": 6.253361701965332, "learning_rate": 5.181818181818182e-07, "num_tokens": 5868309.0, "completions/mean_length": 46.75, "completions/min_length": 43.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.75, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9527664184570312, "rewards/meter/std": 0.029389692470431328, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8257386684417725, "rewards/repeat_soft/std": 0.09025226533412933, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.7580687403678894, "rewards/total_composite/std": 0.03661768138408661, "reward": 0.7580687403678894, "reward_std": 0.03661767765879631, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04754331707954407, "sampling/sampling_logp_difference/max": 2.7940101623535156, "sampling/importance_sampling_ratio/min": 0.0611753985285759, "sampling/importance_sampling_ratio/mean": 1.00124990940094, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17878591641783714, "clip_ratio/low_mean": 0.0052083334885537624, "clip_ratio/low_min": 0.0052083334885537624, "clip_ratio/high_mean": 0.020888741593807936, "clip_ratio/high_max": 0.020888741593807936, "clip_ratio/region_mean": 0.026097075082361698, "reward_total_mean": 0.7580687403678894, "reward_meter_mean": 0.9527664184570312, "reward_meter_std": 0.029389692470431328, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8257386684417725, "reward_repeat_soft_std": 0.09025226533412933, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.7580687403678894, "reward_total_composite_std": 0.03661768138408661} {"timestamp_utc": "2026-04-13T06:11:20Z", "mode": "train", "global_step": 3131, "epoch": 0.31451531893520845, "loss": 0.0325, "grad_norm": 10.830806732177734, "learning_rate": 5.151515151515152e-07, "num_tokens": 5869862.0, "completions/mean_length": 50.125, "completions/min_length": 44.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.125, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.8929994106292725, "rewards/meter/std": 0.22321180999279022, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 1.0, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.7975000143051147, "rewards/judge_quality/std": 0.21492525935173035, "rewards/total_composite/mean": 0.8910996913909912, "rewards/total_composite/std": 0.10709404200315475, "reward": 0.8910996913909912, "reward_std": 0.10709403455257416, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11981494724750519, "sampling/sampling_logp_difference/max": 4.206521034240723, "sampling/importance_sampling_ratio/min": 0.014898109249770641, "sampling/importance_sampling_ratio/mean": 0.9869417548179626, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4316507540643215, "clip_ratio/low_mean": 0.04077221918851137, "clip_ratio/low_min": 0.04077221918851137, "clip_ratio/high_mean": 0.07425691513344646, "clip_ratio/high_max": 0.07425691513344646, "clip_ratio/region_mean": 0.11502913432195783, "reward_total_mean": 0.8910996913909912, "reward_meter_mean": 0.8929994106292725, "reward_meter_std": 0.22321180999279022, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 1.0, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.7975000143051147, "reward_judge_quality_std": 0.21492525935173035, "reward_total_composite_mean": 0.8910996913909912, "reward_total_composite_std": 0.10709404200315475} {"timestamp_utc": "2026-04-13T06:11:26Z", "mode": "train", "global_step": 3132, "epoch": 0.3146157709693621, "loss": 0.0259, "grad_norm": 14.81539535522461, "learning_rate": 5.121212121212121e-07, "num_tokens": 5871707.0, "completions/mean_length": 61.625, "completions/min_length": 52.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.625, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.6468722224235535, "rewards/meter/std": 0.46617987751960754, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9729013442993164, "rewards/repeat_soft/std": 0.011310894973576069, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.7768826484680176, "rewards/total_composite/std": 0.23036833107471466, "reward": 0.7768826484680176, "reward_std": 0.23036833107471466, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09907490015029907, "sampling/sampling_logp_difference/max": 1.583583116531372, "sampling/importance_sampling_ratio/min": 0.20523838698863983, "sampling/importance_sampling_ratio/mean": 0.9998711943626404, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4200417473912239, "clip_ratio/low_mean": 0.021939619444310665, "clip_ratio/low_min": 0.021939619444310665, "clip_ratio/high_mean": 0.04828303353860974, "clip_ratio/high_max": 0.04828303353860974, "clip_ratio/region_mean": 0.07022265298292041, "reward_total_mean": 0.7768826484680176, "reward_meter_mean": 0.6468722224235535, "reward_meter_std": 0.46617987751960754, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9729013442993164, "reward_repeat_soft_std": 0.011310894973576069, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.7768826484680176, "reward_total_composite_std": 0.23036833107471466} {"timestamp_utc": "2026-04-13T06:11:33Z", "mode": "train", "global_step": 3133, "epoch": 0.3147162230035158, "loss": 0.0211, "grad_norm": 6.2822489738464355, "learning_rate": 5.090909090909092e-07, "num_tokens": 5874187.0, "completions/mean_length": 120.0, "completions/min_length": 111.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.0, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.9941681623458862, "rewards/meter/std": 0.004123931284993887, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.902656614780426, "rewards/repeat_soft/std": 0.04437338188290596, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.8248913288116455, "rewards/total_composite/std": 0.028958024457097054, "reward": 0.8248913288116455, "reward_std": 0.028958020731806755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07186611741781235, "sampling/sampling_logp_difference/max": 1.7765569686889648, "sampling/importance_sampling_ratio/min": 0.16921977698802948, "sampling/importance_sampling_ratio/mean": 1.0081969499588013, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4061673767864704, "clip_ratio/low_mean": 0.05882978066802025, "clip_ratio/low_min": 0.05882978066802025, "clip_ratio/high_mean": 0.010869565419852734, "clip_ratio/high_max": 0.010869565419852734, "clip_ratio/region_mean": 0.06969934608787298, "reward_total_mean": 0.8248913288116455, "reward_meter_mean": 0.9941681623458862, "reward_meter_std": 0.004123931284993887, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.902656614780426, "reward_repeat_soft_std": 0.04437338188290596, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.8248913288116455, "reward_total_composite_std": 0.028958024457097054} {"timestamp_utc": "2026-04-13T06:11:39Z", "mode": "train", "global_step": 3134, "epoch": 0.3148166750376695, "loss": 0.0205, "grad_norm": 12.534238815307617, "learning_rate": 5.060606060606061e-07, "num_tokens": 5875729.0, "completions/mean_length": 26.75, "completions/min_length": 25.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.75, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.7661738991737366, "rewards/meter/std": 0.3322599232196808, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.961837112903595, "rewards/repeat_soft/std": 0.0018749026348814368, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7237119674682617, "rewards/total_composite/std": 0.14786481857299805, "reward": 0.7237119674682617, "reward_std": 0.14786481857299805, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09635813534259796, "sampling/sampling_logp_difference/max": 1.412766456604004, "sampling/importance_sampling_ratio/min": 0.24346880614757538, "sampling/importance_sampling_ratio/mean": 1.013493299484253, "sampling/importance_sampling_ratio/max": 1.949036717414856, "entropy": 0.486625537276268, "clip_ratio/low_mean": 0.05082417652010918, "clip_ratio/low_min": 0.05082417652010918, "clip_ratio/high_mean": 0.06660847179591656, "clip_ratio/high_max": 0.06660847179591656, "clip_ratio/region_mean": 0.11743264831602573, "reward_total_mean": 0.7237119674682617, "reward_meter_mean": 0.7661738991737366, "reward_meter_std": 0.3322599232196808, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.961837112903595, "reward_repeat_soft_std": 0.0018749026348814368, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7237119674682617, "reward_total_composite_std": 0.14786481857299805} {"timestamp_utc": "2026-04-13T06:11:46Z", "mode": "train", "global_step": 3135, "epoch": 0.3149171270718232, "loss": 0.0138, "grad_norm": 5.960447311401367, "learning_rate": 5.03030303030303e-07, "num_tokens": 5878231.0, "completions/mean_length": 128.75, "completions/min_length": 120.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.75, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.9907165169715881, "rewards/meter/std": 0.004583810456097126, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9192942380905151, "rewards/repeat_soft/std": 0.03916727378964424, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.7882518172264099, "rewards/total_composite/std": 0.02791648358106613, "reward": 0.7882518172264099, "reward_std": 0.027916481718420982, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0864502415060997, "sampling/sampling_logp_difference/max": 2.0309605598449707, "sampling/importance_sampling_ratio/min": 0.13120941817760468, "sampling/importance_sampling_ratio/mean": 1.0134800672531128, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3996247947216034, "clip_ratio/low_mean": 0.03926136717200279, "clip_ratio/low_min": 0.03926136717200279, "clip_ratio/high_mean": 0.036049772053956985, "clip_ratio/high_max": 0.036049772053956985, "clip_ratio/region_mean": 0.07531113922595978, "reward_total_mean": 0.7882518172264099, "reward_meter_mean": 0.9907165169715881, "reward_meter_std": 0.004583810456097126, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9192942380905151, "reward_repeat_soft_std": 0.03916727378964424, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.7882518172264099, "reward_total_composite_std": 0.02791648358106613} {"timestamp_utc": "2026-04-13T06:11:54Z", "mode": "train", "global_step": 3136, "epoch": 0.3150175791059769, "loss": 0.0231, "grad_norm": 5.851212501525879, "learning_rate": 5.000000000000001e-07, "num_tokens": 5879966.0, "completions/mean_length": 67.875, "completions/min_length": 65.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.875, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9656972885131836, "rewards/meter/std": 0.010491170920431614, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8850535154342651, "rewards/repeat_soft/std": 0.10218026489019394, "rewards/judge_quality/mean": 0.2175000011920929, "rewards/judge_quality/std": 0.1249857097864151, "rewards/total_composite/mean": 0.7383191585540771, "rewards/total_composite/std": 0.04441024735569954, "reward": 0.7383191585540771, "reward_std": 0.04441023990511894, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07092919200658798, "sampling/sampling_logp_difference/max": 2.054126739501953, "sampling/importance_sampling_ratio/min": 0.12820473313331604, "sampling/importance_sampling_ratio/mean": 0.9943328499794006, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31075877137482166, "clip_ratio/low_mean": 0.027892548590898514, "clip_ratio/low_min": 0.027892548590898514, "clip_ratio/high_mean": 0.018278174102306366, "clip_ratio/high_max": 0.018278174102306366, "clip_ratio/region_mean": 0.04617072269320488, "reward_total_mean": 0.7383191585540771, "reward_meter_mean": 0.9656972885131836, "reward_meter_std": 0.010491170920431614, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8850535154342651, "reward_repeat_soft_std": 0.10218026489019394, "reward_judge_quality_mean": 0.2175000011920929, "reward_judge_quality_std": 0.1249857097864151, "reward_total_composite_mean": 0.7383191585540771, "reward_total_composite_std": 0.04441024735569954} {"timestamp_utc": "2026-04-13T06:12:00Z", "mode": "train", "global_step": 3137, "epoch": 0.3151180311401306, "loss": 0.0034, "grad_norm": 9.260276794433594, "learning_rate": 4.96969696969697e-07, "num_tokens": 5881320.0, "completions/mean_length": 24.25, "completions/min_length": 23.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.25, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.9876021146774292, "rewards/meter/std": 0.0021307957358658314, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9585226774215698, "rewards/repeat_soft/std": 0.011249415576457977, "rewards/judge_quality/mean": 0.42499998211860657, "rewards/judge_quality/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8177732229232788, "rewards/total_composite/std": 0.021256931126117706, "reward": 0.8177732229232788, "reward_std": 0.021256916224956512, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06423301994800568, "sampling/sampling_logp_difference/max": 0.875057578086853, "sampling/importance_sampling_ratio/min": 0.41683802008628845, "sampling/importance_sampling_ratio/mean": 1.0005680322647095, "sampling/importance_sampling_ratio/max": 1.4638712406158447, "entropy": 0.320328064262867, "clip_ratio/low_mean": 0.0052083334885537624, "clip_ratio/low_min": 0.0052083334885537624, "clip_ratio/high_mean": 0.07224839041009545, "clip_ratio/high_max": 0.07224839041009545, "clip_ratio/region_mean": 0.07745672389864922, "reward_total_mean": 0.8177732229232788, "reward_meter_mean": 0.9876021146774292, "reward_meter_std": 0.0021307957358658314, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9585226774215698, "reward_repeat_soft_std": 0.011249415576457977, "reward_judge_quality_mean": 0.42499998211860657, "reward_judge_quality_std": 0.0707106739282608, "reward_total_composite_mean": 0.8177732229232788, "reward_total_composite_std": 0.021256931126117706} {"timestamp_utc": "2026-04-13T06:12:08Z", "mode": "train", "global_step": 3138, "epoch": 0.3152184831742843, "loss": -0.0127, "grad_norm": 6.895051956176758, "learning_rate": 4.93939393939394e-07, "num_tokens": 5883471.0, "completions/mean_length": 105.875, "completions/min_length": 101.0, "completions/max_length": 115.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.875, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 115.0, "rewards/meter/mean": 0.9891356229782104, "rewards/meter/std": 0.009855036623775959, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9435887336730957, "rewards/repeat_soft/std": 0.04187345504760742, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8154699206352234, "rewards/total_composite/std": 0.005982470698654652, "reward": 0.8154699206352234, "reward_std": 0.005982470232993364, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08650524169206619, "sampling/sampling_logp_difference/max": 2.630707263946533, "sampling/importance_sampling_ratio/min": 0.07202750444412231, "sampling/importance_sampling_ratio/mean": 1.0119357109069824, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45607730746269226, "clip_ratio/low_mean": 0.046258973656222224, "clip_ratio/low_min": 0.046258973656222224, "clip_ratio/high_mean": 0.03209282550960779, "clip_ratio/high_max": 0.03209282550960779, "clip_ratio/region_mean": 0.07835179916583002, "reward_total_mean": 0.8154699206352234, "reward_meter_mean": 0.9891356229782104, "reward_meter_std": 0.009855036623775959, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9435887336730957, "reward_repeat_soft_std": 0.04187345504760742, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8154699206352234, "reward_total_composite_std": 0.005982470698654652} {"timestamp_utc": "2026-04-13T06:12:15Z", "mode": "train", "global_step": 3139, "epoch": 0.31531893520843796, "loss": 0.0063, "grad_norm": 3.46791934967041, "learning_rate": 4.909090909090909e-07, "num_tokens": 5885454.0, "completions/mean_length": 83.875, "completions/min_length": 80.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.875, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.9837787747383118, "rewards/meter/std": 0.017334360629320145, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9863717555999756, "rewards/repeat_soft/std": 0.009402680210769176, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.8383376002311707, "rewards/total_composite/std": 0.05413082242012024, "reward": 0.8383376002311707, "reward_std": 0.05413081869482994, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06575880944728851, "sampling/sampling_logp_difference/max": 1.7547407150268555, "sampling/importance_sampling_ratio/min": 0.1729520857334137, "sampling/importance_sampling_ratio/mean": 0.9912378787994385, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2507315520197153, "clip_ratio/low_mean": 0.043049678672105074, "clip_ratio/low_min": 0.043049678672105074, "clip_ratio/high_mean": 0.009036144241690636, "clip_ratio/high_max": 0.009036144241690636, "clip_ratio/region_mean": 0.05208582291379571, "reward_total_mean": 0.8383376002311707, "reward_meter_mean": 0.9837787747383118, "reward_meter_std": 0.017334360629320145, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9863717555999756, "reward_repeat_soft_std": 0.009402680210769176, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.8383376002311707, "reward_total_composite_std": 0.05413082242012024} {"timestamp_utc": "2026-04-13T06:12:22Z", "mode": "train", "global_step": 3140, "epoch": 0.31541938724259166, "loss": 0.0154, "grad_norm": 8.19749641418457, "learning_rate": 4.878787878787879e-07, "num_tokens": 5887449.0, "completions/mean_length": 67.375, "completions/min_length": 61.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.375, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9037306904792786, "rewards/meter/std": 0.22889646887779236, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9916646480560303, "rewards/repeat_soft/std": 0.010159045457839966, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.7788453102111816, "rewards/total_composite/std": 0.10375384986400604, "reward": 0.7788453102111816, "reward_std": 0.10375385731458664, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08441368490457535, "sampling/sampling_logp_difference/max": 1.8139584064483643, "sampling/importance_sampling_ratio/min": 0.16300761699676514, "sampling/importance_sampling_ratio/mean": 1.0107158422470093, "sampling/importance_sampling_ratio/max": 1.993990421295166, "entropy": 0.41371146962046623, "clip_ratio/low_mean": 0.028555141761898994, "clip_ratio/low_min": 0.028555141761898994, "clip_ratio/high_mean": 0.05427739256992936, "clip_ratio/high_max": 0.05427739256992936, "clip_ratio/region_mean": 0.08283253433182836, "reward_total_mean": 0.7788453102111816, "reward_meter_mean": 0.9037306904792786, "reward_meter_std": 0.22889646887779236, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9916646480560303, "reward_repeat_soft_std": 0.010159045457839966, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.7788453102111816, "reward_total_composite_std": 0.10375384986400604} {"timestamp_utc": "2026-04-13T06:12:29Z", "mode": "train", "global_step": 3141, "epoch": 0.3155198392767454, "loss": 0.0341, "grad_norm": 6.688923358917236, "learning_rate": 4.848484848484849e-07, "num_tokens": 5889665.0, "completions/mean_length": 87.0, "completions/min_length": 81.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.0, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.9916439056396484, "rewards/meter/std": 0.0046782647259533405, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9322282075881958, "rewards/repeat_soft/std": 0.04566548392176628, "rewards/judge_quality/mean": 0.5112500190734863, "rewards/judge_quality/std": 0.23061642050743103, "rewards/total_composite/mean": 0.8428375720977783, "rewards/total_composite/std": 0.07086004316806793, "reward": 0.8428375720977783, "reward_std": 0.07086003571748734, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08454222232103348, "sampling/sampling_logp_difference/max": 2.8479256629943848, "sampling/importance_sampling_ratio/min": 0.05796443670988083, "sampling/importance_sampling_ratio/mean": 0.9928715229034424, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36085645854473114, "clip_ratio/low_mean": 0.0511031576897949, "clip_ratio/low_min": 0.0511031576897949, "clip_ratio/high_mean": 0.017659231089055538, "clip_ratio/high_max": 0.017659231089055538, "clip_ratio/region_mean": 0.06876238877885044, "reward_total_mean": 0.8428375720977783, "reward_meter_mean": 0.9916439056396484, "reward_meter_std": 0.0046782647259533405, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9322282075881958, "reward_repeat_soft_std": 0.04566548392176628, "reward_judge_quality_mean": 0.5112500190734863, "reward_judge_quality_std": 0.23061642050743103, "reward_total_composite_mean": 0.8428375720977783, "reward_total_composite_std": 0.07086004316806793} {"timestamp_utc": "2026-04-13T06:12:35Z", "mode": "train", "global_step": 3142, "epoch": 0.315620291310899, "loss": 0.0222, "grad_norm": 6.618165493011475, "learning_rate": 4.818181818181818e-07, "num_tokens": 5891706.0, "completions/mean_length": 77.125, "completions/min_length": 73.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.125, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9747964143753052, "rewards/meter/std": 0.03850504010915756, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9365788698196411, "rewards/repeat_soft/std": 0.023552922531962395, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.8206912279129028, "rewards/total_composite/std": 0.0394861213862896, "reward": 0.8206912279129028, "reward_std": 0.039486128836870193, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09234880656003952, "sampling/sampling_logp_difference/max": 1.3988423347473145, "sampling/importance_sampling_ratio/min": 0.2468826174736023, "sampling/importance_sampling_ratio/mean": 1.0073233842849731, "sampling/importance_sampling_ratio/max": 1.8587474822998047, "entropy": 0.4222510755062103, "clip_ratio/low_mean": 0.06678069615736604, "clip_ratio/low_min": 0.06678069615736604, "clip_ratio/high_mean": 0.02483933698385954, "clip_ratio/high_max": 0.02483933698385954, "clip_ratio/region_mean": 0.09162003314122558, "reward_total_mean": 0.8206912279129028, "reward_meter_mean": 0.9747964143753052, "reward_meter_std": 0.03850504010915756, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9365788698196411, "reward_repeat_soft_std": 0.023552922531962395, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.8206912279129028, "reward_total_composite_std": 0.0394861213862896} {"timestamp_utc": "2026-04-13T06:12:42Z", "mode": "train", "global_step": 3143, "epoch": 0.31572074334505273, "loss": 0.0038, "grad_norm": 17.878459930419922, "learning_rate": 4.787878787878789e-07, "num_tokens": 5893321.0, "completions/mean_length": 32.875, "completions/min_length": 31.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.875, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9946584701538086, "rewards/meter/std": 0.002423071302473545, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9610613584518433, "rewards/repeat_soft/std": 0.004068940877914429, "rewards/judge_quality/mean": 0.44624999165534973, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8275774717330933, "rewards/total_composite/std": 0.0030167195945978165, "reward": 0.8275774717330933, "reward_std": 0.0030167237855494022, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11069350689649582, "sampling/sampling_logp_difference/max": 2.2654411792755127, "sampling/importance_sampling_ratio/min": 0.1037842333316803, "sampling/importance_sampling_ratio/mean": 0.9910959601402283, "sampling/importance_sampling_ratio/max": 1.8832039833068848, "entropy": 0.5071985460817814, "clip_ratio/low_mean": 0.01785714365541935, "clip_ratio/low_min": 0.01785714365541935, "clip_ratio/high_mean": 0.07980510871857405, "clip_ratio/high_max": 0.07980510871857405, "clip_ratio/region_mean": 0.0976622523739934, "reward_total_mean": 0.8275774717330933, "reward_meter_mean": 0.9946584701538086, "reward_meter_std": 0.002423071302473545, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9610613584518433, "reward_repeat_soft_std": 0.004068940877914429, "reward_judge_quality_mean": 0.44624999165534973, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8275774717330933, "reward_total_composite_std": 0.0030167195945978165} {"timestamp_utc": "2026-04-13T06:12:49Z", "mode": "train", "global_step": 3144, "epoch": 0.31582119537920644, "loss": 0.0131, "grad_norm": 8.721766471862793, "learning_rate": 4.757575757575758e-07, "num_tokens": 5895306.0, "completions/mean_length": 72.125, "completions/min_length": 65.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.125, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9889914989471436, "rewards/meter/std": 0.0034819927532225847, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9737226963043213, "rewards/repeat_soft/std": 0.028003137558698654, "rewards/judge_quality/mean": 0.4312499761581421, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8217934370040894, "rewards/total_composite/std": 0.00494422996416688, "reward": 0.8217934370040894, "reward_std": 0.004944223444908857, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07880930602550507, "sampling/sampling_logp_difference/max": 2.377230644226074, "sampling/importance_sampling_ratio/min": 0.31196537613868713, "sampling/importance_sampling_ratio/mean": 1.0039949417114258, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3295420780777931, "clip_ratio/low_mean": 0.02059800166171044, "clip_ratio/low_min": 0.02059800166171044, "clip_ratio/high_mean": 0.04486945550888777, "clip_ratio/high_max": 0.04486945550888777, "clip_ratio/region_mean": 0.06546745717059821, "reward_total_mean": 0.8217934370040894, "reward_meter_mean": 0.9889914989471436, "reward_meter_std": 0.0034819927532225847, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9737226963043213, "reward_repeat_soft_std": 0.028003137558698654, "reward_judge_quality_mean": 0.4312499761581421, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8217934370040894, "reward_total_composite_std": 0.00494422996416688} {"timestamp_utc": "2026-04-13T06:12:56Z", "mode": "train", "global_step": 3145, "epoch": 0.3159216474133601, "loss": 0.0069, "grad_norm": 7.130085468292236, "learning_rate": 4.7272727272727273e-07, "num_tokens": 5897625.0, "completions/mean_length": 115.875, "completions/min_length": 110.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.875, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.7511761784553528, "rewards/meter/std": 0.30082520842552185, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8986175656318665, "rewards/repeat_soft/std": 0.04190993681550026, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7038910388946533, "rewards/total_composite/std": 0.13688403367996216, "reward": 0.7038910388946533, "reward_std": 0.13688403367996216, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08004676550626755, "sampling/sampling_logp_difference/max": 2.8255860805511475, "sampling/importance_sampling_ratio/min": 0.05927390605211258, "sampling/importance_sampling_ratio/mean": 0.9979087114334106, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3756712004542351, "clip_ratio/low_mean": 0.0077782131265848875, "clip_ratio/low_min": 0.0077782131265848875, "clip_ratio/high_mean": 0.0695788417942822, "clip_ratio/high_max": 0.0695788417942822, "clip_ratio/region_mean": 0.07735705492086709, "reward_total_mean": 0.7038910388946533, "reward_meter_mean": 0.7511761784553528, "reward_meter_std": 0.30082520842552185, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8986175656318665, "reward_repeat_soft_std": 0.04190993681550026, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7038910388946533, "reward_total_composite_std": 0.13688403367996216} {"timestamp_utc": "2026-04-13T06:13:04Z", "mode": "train", "global_step": 3146, "epoch": 0.3160220994475138, "loss": 0.0206, "grad_norm": 6.215914726257324, "learning_rate": 4.696969696969697e-07, "num_tokens": 5900568.0, "completions/mean_length": 167.875, "completions/min_length": 159.0, "completions/max_length": 180.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 167.875, "completions/min_terminated_length": 159.0, "completions/max_terminated_length": 180.0, "rewards/meter/mean": 0.9898424744606018, "rewards/meter/std": 0.005625097546726465, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9361006021499634, "rewards/repeat_soft/std": 0.025754591450095177, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.804914116859436, "rewards/total_composite/std": 0.0189509280025959, "reward": 0.804914116859436, "reward_std": 0.01895092986524105, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0851580873131752, "sampling/sampling_logp_difference/max": 1.8033593893051147, "sampling/importance_sampling_ratio/min": 0.1647445261478424, "sampling/importance_sampling_ratio/mean": 1.0017415285110474, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.38619448989629745, "clip_ratio/low_mean": 0.018791554495692253, "clip_ratio/low_min": 0.018791554495692253, "clip_ratio/high_mean": 0.07524016825482249, "clip_ratio/high_max": 0.07524016825482249, "clip_ratio/region_mean": 0.09403172275051475, "reward_total_mean": 0.804914116859436, "reward_meter_mean": 0.9898424744606018, "reward_meter_std": 0.005625097546726465, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9361006021499634, "reward_repeat_soft_std": 0.025754591450095177, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.804914116859436, "reward_total_composite_std": 0.0189509280025959} {"timestamp_utc": "2026-04-13T06:13:12Z", "mode": "train", "global_step": 3147, "epoch": 0.3161225514816675, "loss": 0.096, "grad_norm": 6.386839866638184, "learning_rate": 4.666666666666667e-07, "num_tokens": 5903239.0, "completions/mean_length": 150.875, "completions/min_length": 130.0, "completions/max_length": 177.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 150.875, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 177.0, "rewards/meter/mean": 0.6211768388748169, "rewards/meter/std": 0.26186275482177734, "rewards/count_adherence/mean": 0.9166666269302368, "rewards/count_adherence/std": 0.0890870913863182, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9204990863800049, "rewards/repeat_soft/std": 0.03999796137213707, "rewards/judge_quality/mean": 0.3474999964237213, "rewards/judge_quality/std": 0.1895671933889389, "rewards/total_composite/mean": 0.6133294701576233, "rewards/total_composite/std": 0.12335503846406937, "reward": 0.6133294701576233, "reward_std": 0.12335504591464996, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10895524173974991, "sampling/sampling_logp_difference/max": 2.9509806632995605, "sampling/importance_sampling_ratio/min": 0.05228840187191963, "sampling/importance_sampling_ratio/mean": 0.9924812316894531, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4594745896756649, "clip_ratio/low_mean": 0.04599835351109505, "clip_ratio/low_min": 0.04599835351109505, "clip_ratio/high_mean": 0.05655346438288689, "clip_ratio/high_max": 0.05655346438288689, "clip_ratio/region_mean": 0.10255181789398193, "reward_total_mean": 0.6133294701576233, "reward_meter_mean": 0.6211768388748169, "reward_meter_std": 0.26186275482177734, "reward_count_adherence_mean": 0.9166666269302368, "reward_count_adherence_std": 0.0890870913863182, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9204990863800049, "reward_repeat_soft_std": 0.03999796137213707, "reward_judge_quality_mean": 0.3474999964237213, "reward_judge_quality_std": 0.1895671933889389, "reward_total_composite_mean": 0.6133294701576233, "reward_total_composite_std": 0.12335503846406937} {"timestamp_utc": "2026-04-13T06:13:19Z", "mode": "train", "global_step": 3148, "epoch": 0.3162230035158212, "loss": 0.0512, "grad_norm": 15.15209674835205, "learning_rate": 4.6363636363636365e-07, "num_tokens": 5904826.0, "completions/mean_length": 52.375, "completions/min_length": 50.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.375, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9867373704910278, "rewards/meter/std": 0.001777186756953597, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9619555473327637, "rewards/repeat_soft/std": 0.03451661020517349, "rewards/judge_quality/mean": 0.42124998569488525, "rewards/judge_quality/std": 0.06998724490404129, "rewards/total_composite/mean": 0.8166024088859558, "rewards/total_composite/std": 0.022787440568208694, "reward": 0.8166024088859558, "reward_std": 0.022787438705563545, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07478220015764236, "sampling/sampling_logp_difference/max": 1.2120085954666138, "sampling/importance_sampling_ratio/min": 0.2975989282131195, "sampling/importance_sampling_ratio/mean": 1.0079253911972046, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3479553870856762, "clip_ratio/low_mean": 0.011597503442317247, "clip_ratio/low_min": 0.011597503442317247, "clip_ratio/high_mean": 0.06783119728788733, "clip_ratio/high_max": 0.06783119728788733, "clip_ratio/region_mean": 0.07942870073020458, "reward_total_mean": 0.8166024088859558, "reward_meter_mean": 0.9867373704910278, "reward_meter_std": 0.001777186756953597, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9619555473327637, "reward_repeat_soft_std": 0.03451661020517349, "reward_judge_quality_mean": 0.42124998569488525, "reward_judge_quality_std": 0.06998724490404129, "reward_total_composite_mean": 0.8166024088859558, "reward_total_composite_std": 0.022787440568208694} {"timestamp_utc": "2026-04-13T06:13:25Z", "mode": "train", "global_step": 3149, "epoch": 0.3163234555499749, "loss": 0.0485, "grad_norm": 13.485732078552246, "learning_rate": 4.6060606060606064e-07, "num_tokens": 5906313.0, "completions/mean_length": 29.875, "completions/min_length": 27.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.875, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9888427257537842, "rewards/meter/std": 0.004514485131949186, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9594928026199341, "rewards/repeat_soft/std": 0.008505655452609062, "rewards/judge_quality/mean": 0.36375001072883606, "rewards/judge_quality/std": 0.13265825808048248, "rewards/total_composite/mean": 0.8000534772872925, "rewards/total_composite/std": 0.039709337055683136, "reward": 0.8000534772872925, "reward_std": 0.03970934450626373, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10819368809461594, "sampling/sampling_logp_difference/max": 1.330676794052124, "sampling/importance_sampling_ratio/min": 0.26429834961891174, "sampling/importance_sampling_ratio/mean": 0.9806923270225525, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3988439105451107, "clip_ratio/low_mean": 0.011844757944345474, "clip_ratio/low_min": 0.011844757944345474, "clip_ratio/high_mean": 0.07685700245201588, "clip_ratio/high_max": 0.07685700245201588, "clip_ratio/region_mean": 0.08870176039636135, "reward_total_mean": 0.8000534772872925, "reward_meter_mean": 0.9888427257537842, "reward_meter_std": 0.004514485131949186, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9594928026199341, "reward_repeat_soft_std": 0.008505655452609062, "reward_judge_quality_mean": 0.36375001072883606, "reward_judge_quality_std": 0.13265825808048248, "reward_total_composite_mean": 0.8000534772872925, "reward_total_composite_std": 0.039709337055683136} {"timestamp_utc": "2026-04-13T06:13:36Z", "mode": "train", "global_step": 3150, "epoch": 0.3164239075841286, "loss": -0.1548, "grad_norm": 1.303343653678894, "learning_rate": 4.5757575757575764e-07, "num_tokens": 5908051.0, "completions/mean_length": 115.25, "completions/min_length": 57.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 58.57143020629883, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9234009981155396, "rewards/meter/std": 0.18976329267024994, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.901565432548523, "rewards/repeat_soft/std": 0.03684718906879425, "rewards/judge_quality/mean": 0.3387500047683716, "rewards/judge_quality/std": 0.14327171444892883, "rewards/total_composite/mean": 0.6991443037986755, "rewards/total_composite/std": 0.2835082709789276, "reward": 0.6991443037986755, "reward_std": 0.2835082411766052, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06120269000530243, "sampling/sampling_logp_difference/max": 1.0380223989486694, "sampling/importance_sampling_ratio/min": 0.3541543781757355, "sampling/importance_sampling_ratio/mean": 1.0113359689712524, "sampling/importance_sampling_ratio/max": 1.867425799369812, "entropy": 0.3081495948135853, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.05749358213506639, "clip_ratio/high_max": 0.05749358213506639, "clip_ratio/region_mean": 0.05749358213506639, "reward_total_mean": 0.6991443037986755, "reward_meter_mean": 0.9234009981155396, "reward_meter_std": 0.18976329267024994, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.901565432548523, "reward_repeat_soft_std": 0.03684718906879425, "reward_judge_quality_mean": 0.3387500047683716, "reward_judge_quality_std": 0.14327171444892883, "reward_total_composite_mean": 0.6991443037986755, "reward_total_composite_std": 0.2835082709789276} {"timestamp_utc": "2026-04-13T06:14:35Z", "mode": "eval", "global_step": 3150, "epoch": 0.3164239075841286, "eval_loss": NaN, "eval_runtime": 58.5986, "eval_samples_per_second": 1.365, "eval_steps_per_second": 0.171, "eval_num_tokens": 5908051.0, "eval_completions/mean_length": 118.675, "eval_completions/min_length": 42.7, "eval_completions/max_length": 272.7, "eval_completions/clipped_ratio": 0.05, "eval_completions/mean_terminated_length": 97.9422622680664, "eval_completions/min_terminated_length": 42.7, "eval_completions/max_terminated_length": 161.9, "eval_rewards/meter/mean": 0.9119099020957947, "eval_rewards/meter/std": 0.14346593022346496, "eval_rewards/count_adherence/mean": 0.9866666615009307, "eval_rewards/count_adherence/std": 0.026050623133778573, "eval_rewards/hard_gate/mean": 0.9375, "eval_rewards/hard_gate/std": 0.15235702097415924, "eval_rewards/repeat_soft/mean": 0.9013808310031891, "eval_rewards/repeat_soft/std": 0.08577343747019768, "eval_rewards/judge_quality/mean": 0.42062499225139616, "eval_rewards/judge_quality/std": 0.16210326803848146, "eval_rewards/total_composite/mean": 0.7381643891334534, "eval_rewards/total_composite/std": 0.16528972573578357, "eval_reward": 0.7381643891334534, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.0338812168687582, "eval_sampling/sampling_logp_difference/max": 0.7283513307571411, "eval_sampling/importance_sampling_ratio/min": 0.4908097118139267, "eval_sampling/importance_sampling_ratio/mean": 1.0070852756500244, "eval_sampling/importance_sampling_ratio/max": 1.2804295301437378, "eval_entropy": 0.3444785088300705, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7381643891334534, "eval_reward_meter_mean": 0.9119099020957947, "eval_reward_meter_std": 0.14346593022346496, "eval_reward_count_adherence_mean": 0.9866666615009307, "eval_reward_count_adherence_std": 0.026050623133778573, "eval_reward_hard_gate_mean": 0.9375, "eval_reward_hard_gate_std": 0.15235702097415924, "eval_reward_repeat_soft_mean": 0.9013808310031891, "eval_reward_repeat_soft_std": 0.08577343747019768, "eval_reward_judge_quality_mean": 0.42062499225139616, "eval_reward_judge_quality_std": 0.16210326803848146, "eval_reward_total_composite_mean": 0.7381643891334534, "eval_reward_total_composite_std": 0.16528972573578357} {"timestamp_utc": "2026-04-13T06:14:44Z", "mode": "train", "global_step": 3151, "epoch": 0.3165243596182823, "loss": -0.0031, "grad_norm": 7.385093688964844, "learning_rate": 4.5454545454545457e-07, "num_tokens": 5909987.0, "completions/mean_length": 82.0, "completions/min_length": 73.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.0, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9776579141616821, "rewards/meter/std": 0.020267615094780922, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9808753728866577, "rewards/repeat_soft/std": 0.014173904433846474, "rewards/judge_quality/mean": 0.5699999928474426, "rewards/judge_quality/std": 0.16035676002502441, "rewards/total_composite/mean": 0.8590335845947266, "rewards/total_composite/std": 0.04465195909142494, "reward": 0.8590335845947266, "reward_std": 0.044651951640844345, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08366034179925919, "sampling/sampling_logp_difference/max": 1.3908252716064453, "sampling/importance_sampling_ratio/min": 0.24886983633041382, "sampling/importance_sampling_ratio/mean": 1.0186463594436646, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4358726777136326, "clip_ratio/low_mean": 0.03663691505789757, "clip_ratio/low_min": 0.03663691505789757, "clip_ratio/high_mean": 0.03455996187403798, "clip_ratio/high_max": 0.03455996187403798, "clip_ratio/region_mean": 0.07119687693193555, "reward_total_mean": 0.8590335845947266, "reward_meter_mean": 0.9776579141616821, "reward_meter_std": 0.020267615094780922, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9808753728866577, "reward_repeat_soft_std": 0.014173904433846474, "reward_judge_quality_mean": 0.5699999928474426, "reward_judge_quality_std": 0.16035676002502441, "reward_total_composite_mean": 0.8590335845947266, "reward_total_composite_std": 0.04465195909142494} {"timestamp_utc": "2026-04-13T06:14:51Z", "mode": "train", "global_step": 3152, "epoch": 0.31662481165243594, "loss": 0.0184, "grad_norm": 10.164688110351562, "learning_rate": 4.5151515151515156e-07, "num_tokens": 5911595.0, "completions/mean_length": 31.0, "completions/min_length": 28.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9938604831695557, "rewards/meter/std": 0.0019623839762061834, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8881510496139526, "rewards/repeat_soft/std": 0.0887170284986496, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.81767737865448, "rewards/total_composite/std": 0.008436748757958412, "reward": 0.81767737865448, "reward_std": 0.00843674223870039, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08744953572750092, "sampling/sampling_logp_difference/max": 1.4250540733337402, "sampling/importance_sampling_ratio/min": 0.2404954731464386, "sampling/importance_sampling_ratio/mean": 1.002399206161499, "sampling/importance_sampling_ratio/max": 1.6441481113433838, "entropy": 0.39472753182053566, "clip_ratio/low_mean": 0.012834821827709675, "clip_ratio/low_min": 0.012834821827709675, "clip_ratio/high_mean": 0.048269910737872124, "clip_ratio/high_max": 0.048269910737872124, "clip_ratio/region_mean": 0.0611047325655818, "reward_total_mean": 0.81767737865448, "reward_meter_mean": 0.9938604831695557, "reward_meter_std": 0.0019623839762061834, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8881510496139526, "reward_repeat_soft_std": 0.0887170284986496, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.81767737865448, "reward_total_composite_std": 0.008436748757958412} {"timestamp_utc": "2026-04-13T06:14:58Z", "mode": "train", "global_step": 3153, "epoch": 0.31672526368658965, "loss": 0.0242, "grad_norm": 6.6839118003845215, "learning_rate": 4.484848484848485e-07, "num_tokens": 5913508.0, "completions/mean_length": 81.125, "completions/min_length": 73.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.125, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.993329644203186, "rewards/meter/std": 0.007509894669055939, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9320617914199829, "rewards/repeat_soft/std": 0.04733073711395264, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.853704571723938, "rewards/total_composite/std": 0.06840092688798904, "reward": 0.853704571723938, "reward_std": 0.06840093433856964, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08369644731283188, "sampling/sampling_logp_difference/max": 4.338709354400635, "sampling/importance_sampling_ratio/min": 0.013053365051746368, "sampling/importance_sampling_ratio/mean": 1.0023658275604248, "sampling/importance_sampling_ratio/max": 1.9249343872070312, "entropy": 0.40419353544712067, "clip_ratio/low_mean": 0.062006679363548756, "clip_ratio/low_min": 0.062006679363548756, "clip_ratio/high_mean": 0.017482943832874298, "clip_ratio/high_max": 0.017482943832874298, "clip_ratio/region_mean": 0.07948962319642305, "reward_total_mean": 0.853704571723938, "reward_meter_mean": 0.993329644203186, "reward_meter_std": 0.007509894669055939, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9320617914199829, "reward_repeat_soft_std": 0.04733073711395264, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.853704571723938, "reward_total_composite_std": 0.06840092688798904} {"timestamp_utc": "2026-04-13T06:15:06Z", "mode": "train", "global_step": 3154, "epoch": 0.31682571572074336, "loss": 0.006, "grad_norm": 5.614310264587402, "learning_rate": 4.454545454545455e-07, "num_tokens": 5916247.0, "completions/mean_length": 152.375, "completions/min_length": 146.0, "completions/max_length": 162.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 152.375, "completions/min_terminated_length": 146.0, "completions/max_terminated_length": 162.0, "rewards/meter/mean": 0.9662624597549438, "rewards/meter/std": 0.07682525366544724, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9497745037078857, "rewards/repeat_soft/std": 0.03133668377995491, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.8432955741882324, "rewards/total_composite/std": 0.08335129171609879, "reward": 0.8432955741882324, "reward_std": 0.08335127681493759, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07820155471563339, "sampling/sampling_logp_difference/max": 1.6063427925109863, "sampling/importance_sampling_ratio/min": 0.20061999559402466, "sampling/importance_sampling_ratio/mean": 1.005393385887146, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3827868476510048, "clip_ratio/low_mean": 0.049538299441337585, "clip_ratio/low_min": 0.049538299441337585, "clip_ratio/high_mean": 0.02040123427286744, "clip_ratio/high_max": 0.02040123427286744, "clip_ratio/region_mean": 0.06993953371420503, "reward_total_mean": 0.8432955741882324, "reward_meter_mean": 0.9662624597549438, "reward_meter_std": 0.07682525366544724, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9497745037078857, "reward_repeat_soft_std": 0.03133668377995491, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.8432955741882324, "reward_total_composite_std": 0.08335129171609879} {"timestamp_utc": "2026-04-13T06:15:12Z", "mode": "train", "global_step": 3155, "epoch": 0.316926167754897, "loss": -0.0094, "grad_norm": 7.114436149597168, "learning_rate": 4.424242424242425e-07, "num_tokens": 5917716.0, "completions/mean_length": 44.625, "completions/min_length": 41.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.625, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9499830603599548, "rewards/meter/std": 0.03710728511214256, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9157963991165161, "rewards/repeat_soft/std": 0.07976768910884857, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.793196976184845, "rewards/total_composite/std": 0.03719424828886986, "reward": 0.793196976184845, "reward_std": 0.03719425946474075, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05286196619272232, "sampling/sampling_logp_difference/max": 1.11979341506958, "sampling/importance_sampling_ratio/min": 0.3263472020626068, "sampling/importance_sampling_ratio/mean": 1.0012434720993042, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31274309754371643, "clip_ratio/low_mean": 0.028363765217363834, "clip_ratio/low_min": 0.028363765217363834, "clip_ratio/high_mean": 0.030715975211933255, "clip_ratio/high_max": 0.030715975211933255, "clip_ratio/region_mean": 0.05907974042929709, "reward_total_mean": 0.793196976184845, "reward_meter_mean": 0.9499830603599548, "reward_meter_std": 0.03710728511214256, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9157963991165161, "reward_repeat_soft_std": 0.07976768910884857, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.793196976184845, "reward_total_composite_std": 0.03719424828886986} {"timestamp_utc": "2026-04-13T06:15:19Z", "mode": "train", "global_step": 3156, "epoch": 0.3170266197890507, "loss": 0.029, "grad_norm": 7.834848403930664, "learning_rate": 4.393939393939394e-07, "num_tokens": 5919787.0, "completions/mean_length": 98.875, "completions/min_length": 94.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.875, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.9520876407623291, "rewards/meter/std": 0.08856946229934692, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9700179100036621, "rewards/repeat_soft/std": 0.013640745542943478, "rewards/judge_quality/mean": 0.5562500357627869, "rewards/judge_quality/std": 0.31717222929000854, "rewards/total_composite/mean": 0.842316210269928, "rewards/total_composite/std": 0.08574382215738297, "reward": 0.842316210269928, "reward_std": 0.08574379235506058, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10810746252536774, "sampling/sampling_logp_difference/max": 3.9801392555236816, "sampling/importance_sampling_ratio/min": 0.018683038651943207, "sampling/importance_sampling_ratio/mean": 0.9929763674736023, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39159686490893364, "clip_ratio/low_mean": 0.05929862102493644, "clip_ratio/low_min": 0.05929862102493644, "clip_ratio/high_mean": 0.036586856469511986, "clip_ratio/high_max": 0.036586856469511986, "clip_ratio/region_mean": 0.09588547749444842, "reward_total_mean": 0.842316210269928, "reward_meter_mean": 0.9520876407623291, "reward_meter_std": 0.08856946229934692, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9700179100036621, "reward_repeat_soft_std": 0.013640745542943478, "reward_judge_quality_mean": 0.5562500357627869, "reward_judge_quality_std": 0.31717222929000854, "reward_total_composite_mean": 0.842316210269928, "reward_total_composite_std": 0.08574382215738297} {"timestamp_utc": "2026-04-13T06:15:30Z", "mode": "train", "global_step": 3157, "epoch": 0.31712707182320443, "loss": -0.0329, "grad_norm": 16.553693771362305, "learning_rate": 4.363636363636364e-07, "num_tokens": 5921391.0, "completions/mean_length": 42.5, "completions/min_length": 38.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.5, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.646031379699707, "rewards/meter/std": 0.4033779501914978, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9988089799880981, "rewards/repeat_soft/std": 0.001709949690848589, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.7228449583053589, "rewards/total_composite/std": 0.16847455501556396, "reward": 0.7228449583053589, "reward_std": 0.16847455501556396, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10942638665437698, "sampling/sampling_logp_difference/max": 1.5474365949630737, "sampling/importance_sampling_ratio/min": 0.21279273927211761, "sampling/importance_sampling_ratio/mean": 1.010516881942749, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4992419704794884, "clip_ratio/low_mean": 0.015109890373423696, "clip_ratio/low_min": 0.015109890373423696, "clip_ratio/high_mean": 0.0505244517698884, "clip_ratio/high_max": 0.0505244517698884, "clip_ratio/region_mean": 0.0656343421433121, "reward_total_mean": 0.7228449583053589, "reward_meter_mean": 0.646031379699707, "reward_meter_std": 0.4033779501914978, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9988089799880981, "reward_repeat_soft_std": 0.001709949690848589, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.7228449583053589, "reward_total_composite_std": 0.16847455501556396} {"timestamp_utc": "2026-04-13T06:15:38Z", "mode": "train", "global_step": 3158, "epoch": 0.3172275238573581, "loss": -0.0106, "grad_norm": 11.582700729370117, "learning_rate": 4.333333333333334e-07, "num_tokens": 5923048.0, "completions/mean_length": 53.125, "completions/min_length": 47.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.125, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.6927434206008911, "rewards/meter/std": 0.3495880365371704, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.958094596862793, "rewards/repeat_soft/std": 0.041996750980615616, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.6880439519882202, "rewards/total_composite/std": 0.15684442222118378, "reward": 0.6880439519882202, "reward_std": 0.15684442222118378, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08472320437431335, "sampling/sampling_logp_difference/max": 1.5720022916793823, "sampling/importance_sampling_ratio/min": 0.2076290249824524, "sampling/importance_sampling_ratio/mean": 0.9984884262084961, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4118208158761263, "clip_ratio/low_mean": 0.02719647390767932, "clip_ratio/low_min": 0.02719647390767932, "clip_ratio/high_mean": 0.023636832367628813, "clip_ratio/high_max": 0.023636832367628813, "clip_ratio/region_mean": 0.05083330627530813, "reward_total_mean": 0.6880439519882202, "reward_meter_mean": 0.6927434206008911, "reward_meter_std": 0.3495880365371704, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.958094596862793, "reward_repeat_soft_std": 0.041996750980615616, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.6880439519882202, "reward_total_composite_std": 0.15684442222118378} {"timestamp_utc": "2026-04-13T06:15:47Z", "mode": "train", "global_step": 3159, "epoch": 0.3173279758915118, "loss": 0.0175, "grad_norm": 5.645223140716553, "learning_rate": 4.3030303030303034e-07, "num_tokens": 5925498.0, "completions/mean_length": 130.25, "completions/min_length": 125.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 130.25, "completions/min_terminated_length": 125.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.9929522275924683, "rewards/meter/std": 0.0026101868133991957, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7948687672615051, "rewards/repeat_soft/std": 0.10386699438095093, "rewards/judge_quality/mean": 0.47749996185302734, "rewards/judge_quality/std": 0.16263456642627716, "rewards/total_composite/mean": 0.8195653557777405, "rewards/total_composite/std": 0.0494811087846756, "reward": 0.8195653557777405, "reward_std": 0.049481093883514404, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06665202230215073, "sampling/sampling_logp_difference/max": 1.8642370700836182, "sampling/importance_sampling_ratio/min": 0.15501444041728973, "sampling/importance_sampling_ratio/mean": 0.9985721707344055, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.32629815861582756, "clip_ratio/low_mean": 0.054589965380728245, "clip_ratio/low_min": 0.054589965380728245, "clip_ratio/high_mean": 0.0029069767333567142, "clip_ratio/high_max": 0.0029069767333567142, "clip_ratio/region_mean": 0.05749694211408496, "reward_total_mean": 0.8195653557777405, "reward_meter_mean": 0.9929522275924683, "reward_meter_std": 0.0026101868133991957, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7948687672615051, "reward_repeat_soft_std": 0.10386699438095093, "reward_judge_quality_mean": 0.47749996185302734, "reward_judge_quality_std": 0.16263456642627716, "reward_total_composite_mean": 0.8195653557777405, "reward_total_composite_std": 0.0494811087846756} {"timestamp_utc": "2026-04-13T06:15:55Z", "mode": "train", "global_step": 3160, "epoch": 0.3174284279256655, "loss": -0.0485, "grad_norm": 10.262664794921875, "learning_rate": 4.272727272727273e-07, "num_tokens": 5927270.0, "completions/mean_length": 51.5, "completions/min_length": 40.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.5, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9644995927810669, "rewards/meter/std": 0.06065495312213898, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9086437225341797, "rewards/repeat_soft/std": 0.02097679115831852, "rewards/judge_quality/mean": 0.4762499928474426, "rewards/judge_quality/std": 0.09941796213388443, "rewards/total_composite/mean": 0.817764163017273, "rewards/total_composite/std": 0.04561852663755417, "reward": 0.817764163017273, "reward_std": 0.045618534088134766, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09486811608076096, "sampling/sampling_logp_difference/max": 1.6245059967041016, "sampling/importance_sampling_ratio/min": 0.19700898230075836, "sampling/importance_sampling_ratio/mean": 1.0069524049758911, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4816819429397583, "clip_ratio/low_mean": 0.04988503362983465, "clip_ratio/low_min": 0.04988503362983465, "clip_ratio/high_mean": 0.02414532471448183, "clip_ratio/high_max": 0.02414532471448183, "clip_ratio/region_mean": 0.07403035834431648, "reward_total_mean": 0.817764163017273, "reward_meter_mean": 0.9644995927810669, "reward_meter_std": 0.06065495312213898, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9086437225341797, "reward_repeat_soft_std": 0.02097679115831852, "reward_judge_quality_mean": 0.4762499928474426, "reward_judge_quality_std": 0.09941796213388443, "reward_total_composite_mean": 0.817764163017273, "reward_total_composite_std": 0.04561852663755417} {"timestamp_utc": "2026-04-13T06:16:03Z", "mode": "train", "global_step": 3161, "epoch": 0.3175288799598192, "loss": 0.0296, "grad_norm": 5.930389881134033, "learning_rate": 4.242424242424243e-07, "num_tokens": 5929796.0, "completions/mean_length": 127.75, "completions/min_length": 111.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.75, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.9728623032569885, "rewards/meter/std": 0.04102794826030731, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9499365091323853, "rewards/repeat_soft/std": 0.026790175586938858, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.8024067282676697, "rewards/total_composite/std": 0.02287346124649048, "reward": 0.8024067282676697, "reward_std": 0.02287346124649048, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09324408322572708, "sampling/sampling_logp_difference/max": 2.1680757999420166, "sampling/importance_sampling_ratio/min": 0.11439753323793411, "sampling/importance_sampling_ratio/mean": 1.0008655786514282, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4694783687591553, "clip_ratio/low_mean": 0.039558045566082, "clip_ratio/low_min": 0.039558045566082, "clip_ratio/high_mean": 0.056143482215702534, "clip_ratio/high_max": 0.056143482215702534, "clip_ratio/region_mean": 0.09570152778178453, "reward_total_mean": 0.8024067282676697, "reward_meter_mean": 0.9728623032569885, "reward_meter_std": 0.04102794826030731, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9499365091323853, "reward_repeat_soft_std": 0.026790175586938858, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.8024067282676697, "reward_total_composite_std": 0.02287346124649048} {"timestamp_utc": "2026-04-13T06:16:10Z", "mode": "train", "global_step": 3162, "epoch": 0.31762933199397286, "loss": 0.024, "grad_norm": 7.596031188964844, "learning_rate": 4.2121212121212126e-07, "num_tokens": 5931521.0, "completions/mean_length": 60.625, "completions/min_length": 54.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.625, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9918333292007446, "rewards/meter/std": 0.0025873940903693438, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9461057186126709, "rewards/repeat_soft/std": 0.03408718854188919, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.8368105888366699, "rewards/total_composite/std": 0.05276034399867058, "reward": 0.8368105888366699, "reward_std": 0.05276034027338028, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08093980699777603, "sampling/sampling_logp_difference/max": 1.5792236328125, "sampling/importance_sampling_ratio/min": 0.20613506436347961, "sampling/importance_sampling_ratio/mean": 0.9968024492263794, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3773939236998558, "clip_ratio/low_mean": 0.07436302024871111, "clip_ratio/low_min": 0.07436302024871111, "clip_ratio/high_mean": 0.0021929824724793434, "clip_ratio/high_max": 0.0021929824724793434, "clip_ratio/region_mean": 0.07655600272119045, "reward_total_mean": 0.8368105888366699, "reward_meter_mean": 0.9918333292007446, "reward_meter_std": 0.0025873940903693438, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9461057186126709, "reward_repeat_soft_std": 0.03408718854188919, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.8368105888366699, "reward_total_composite_std": 0.05276034399867058} {"timestamp_utc": "2026-04-13T06:16:17Z", "mode": "train", "global_step": 3163, "epoch": 0.31772978402812657, "loss": 0.0829, "grad_norm": 11.268075942993164, "learning_rate": 4.181818181818182e-07, "num_tokens": 5933175.0, "completions/mean_length": 60.75, "completions/min_length": 56.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.75, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9417757987976074, "rewards/meter/std": 0.1164059191942215, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9851474761962891, "rewards/repeat_soft/std": 0.012874401174485683, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8039388656616211, "rewards/total_composite/std": 0.05077788978815079, "reward": 0.8039388656616211, "reward_std": 0.05077788606286049, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09533524513244629, "sampling/sampling_logp_difference/max": 1.2879277467727661, "sampling/importance_sampling_ratio/min": 0.2758418023586273, "sampling/importance_sampling_ratio/mean": 1.0215319395065308, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49744220077991486, "clip_ratio/low_mean": 0.011029412038624287, "clip_ratio/low_min": 0.011029412038624287, "clip_ratio/high_mean": 0.08686249051243067, "clip_ratio/high_max": 0.08686249051243067, "clip_ratio/region_mean": 0.09789190255105495, "reward_total_mean": 0.8039388656616211, "reward_meter_mean": 0.9417757987976074, "reward_meter_std": 0.1164059191942215, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9851474761962891, "reward_repeat_soft_std": 0.012874401174485683, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8039388656616211, "reward_total_composite_std": 0.05077788978815079} {"timestamp_utc": "2026-04-13T06:16:25Z", "mode": "train", "global_step": 3164, "epoch": 0.3178302360622803, "loss": 0.0098, "grad_norm": 6.346017837524414, "learning_rate": 4.1515151515151513e-07, "num_tokens": 5935766.0, "completions/mean_length": 142.875, "completions/min_length": 132.0, "completions/max_length": 150.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.875, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 150.0, "rewards/meter/mean": 0.9950917959213257, "rewards/meter/std": 0.0016323173185810447, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.934804379940033, "rewards/repeat_soft/std": 0.02055746503174305, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7998342514038086, "rewards/total_composite/std": 0.02416626363992691, "reward": 0.7998342514038086, "reward_std": 0.02416626177728176, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08579779416322708, "sampling/sampling_logp_difference/max": 1.1472930908203125, "sampling/importance_sampling_ratio/min": 0.3174950182437897, "sampling/importance_sampling_ratio/mean": 1.0188652276992798, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5416644513607025, "clip_ratio/low_mean": 0.03662979230284691, "clip_ratio/low_min": 0.03662979230284691, "clip_ratio/high_mean": 0.06669889064505696, "clip_ratio/high_max": 0.06669889064505696, "clip_ratio/region_mean": 0.10332868294790387, "reward_total_mean": 0.7998342514038086, "reward_meter_mean": 0.9950917959213257, "reward_meter_std": 0.0016323173185810447, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.934804379940033, "reward_repeat_soft_std": 0.02055746503174305, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7998342514038086, "reward_total_composite_std": 0.02416626363992691} {"timestamp_utc": "2026-04-13T06:16:32Z", "mode": "train", "global_step": 3165, "epoch": 0.31793068809643393, "loss": -0.0003, "grad_norm": 7.850910186767578, "learning_rate": 4.121212121212122e-07, "num_tokens": 5937612.0, "completions/mean_length": 62.75, "completions/min_length": 57.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.75, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9865504503250122, "rewards/meter/std": 0.017065158113837242, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9643900394439697, "rewards/repeat_soft/std": 0.03762424737215042, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.8145117163658142, "rewards/total_composite/std": 0.027053574100136757, "reward": 0.8145117163658142, "reward_std": 0.02705356664955616, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10692285746335983, "sampling/sampling_logp_difference/max": 1.2474117279052734, "sampling/importance_sampling_ratio/min": 0.28724730014801025, "sampling/importance_sampling_ratio/mean": 0.9958828687667847, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5392903313040733, "clip_ratio/low_mean": 0.03454301040619612, "clip_ratio/low_min": 0.03454301040619612, "clip_ratio/high_mean": 0.09687545150518417, "clip_ratio/high_max": 0.09687545150518417, "clip_ratio/region_mean": 0.1314184619113803, "reward_total_mean": 0.8145117163658142, "reward_meter_mean": 0.9865504503250122, "reward_meter_std": 0.017065158113837242, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9643900394439697, "reward_repeat_soft_std": 0.03762424737215042, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.8145117163658142, "reward_total_composite_std": 0.027053574100136757} {"timestamp_utc": "2026-04-13T06:16:39Z", "mode": "train", "global_step": 3166, "epoch": 0.31803114013058764, "loss": 0.0467, "grad_norm": 8.378978729248047, "learning_rate": 4.090909090909091e-07, "num_tokens": 5939264.0, "completions/mean_length": 56.5, "completions/min_length": 48.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.5, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9234935641288757, "rewards/meter/std": 0.1656743884086609, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.976444661617279, "rewards/repeat_soft/std": 0.019667601212859154, "rewards/judge_quality/mean": 0.5724999904632568, "rewards/judge_quality/std": 0.21618445217609406, "rewards/total_composite/mean": 0.8349665403366089, "rewards/total_composite/std": 0.08504947274923325, "reward": 0.8349665403366089, "reward_std": 0.08504949510097504, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09421564638614655, "sampling/sampling_logp_difference/max": 1.4548416137695312, "sampling/importance_sampling_ratio/min": 0.23343732953071594, "sampling/importance_sampling_ratio/mean": 0.9954763650894165, "sampling/importance_sampling_ratio/max": 1.9130990505218506, "entropy": 0.38809067010879517, "clip_ratio/low_mean": 0.0476634562946856, "clip_ratio/low_min": 0.0476634562946856, "clip_ratio/high_mean": 0.018777412362396717, "clip_ratio/high_max": 0.018777412362396717, "clip_ratio/region_mean": 0.06644086865708232, "reward_total_mean": 0.8349665403366089, "reward_meter_mean": 0.9234935641288757, "reward_meter_std": 0.1656743884086609, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.976444661617279, "reward_repeat_soft_std": 0.019667601212859154, "reward_judge_quality_mean": 0.5724999904632568, "reward_judge_quality_std": 0.21618445217609406, "reward_total_composite_mean": 0.8349665403366089, "reward_total_composite_std": 0.08504947274923325} {"timestamp_utc": "2026-04-13T06:16:47Z", "mode": "train", "global_step": 3167, "epoch": 0.31813159216474135, "loss": 0.0632, "grad_norm": 5.528682708740234, "learning_rate": 4.0606060606060605e-07, "num_tokens": 5942045.0, "completions/mean_length": 148.625, "completions/min_length": 136.0, "completions/max_length": 174.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 148.625, "completions/min_terminated_length": 136.0, "completions/max_terminated_length": 174.0, "rewards/meter/mean": 0.9804815053939819, "rewards/meter/std": 0.015509751625359058, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9099512696266174, "rewards/repeat_soft/std": 0.04530250281095505, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.7853367924690247, "rewards/total_composite/std": 0.03654143959283829, "reward": 0.7853367924690247, "reward_std": 0.036541424691677094, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08469509333372116, "sampling/sampling_logp_difference/max": 2.134981870651245, "sampling/importance_sampling_ratio/min": 0.11824672669172287, "sampling/importance_sampling_ratio/mean": 0.9991027116775513, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31651613488793373, "clip_ratio/low_mean": 0.01949973963201046, "clip_ratio/low_min": 0.01949973963201046, "clip_ratio/high_mean": 0.07235544547438622, "clip_ratio/high_max": 0.07235544547438622, "clip_ratio/region_mean": 0.09185518510639668, "reward_total_mean": 0.7853367924690247, "reward_meter_mean": 0.9804815053939819, "reward_meter_std": 0.015509751625359058, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9099512696266174, "reward_repeat_soft_std": 0.04530250281095505, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.7853367924690247, "reward_total_composite_std": 0.03654143959283829} {"timestamp_utc": "2026-04-13T06:16:53Z", "mode": "train", "global_step": 3168, "epoch": 0.318232044198895, "loss": 0.0382, "grad_norm": 9.07286262512207, "learning_rate": 4.030303030303031e-07, "num_tokens": 5943884.0, "completions/mean_length": 63.875, "completions/min_length": 60.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.875, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.6561235189437866, "rewards/meter/std": 0.25921109318733215, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9382648468017578, "rewards/repeat_soft/std": 0.03850472345948219, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.683832049369812, "rewards/total_composite/std": 0.14870306849479675, "reward": 0.683832049369812, "reward_std": 0.14870308339595795, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08921997249126434, "sampling/sampling_logp_difference/max": 1.6106595993041992, "sampling/importance_sampling_ratio/min": 0.19975581765174866, "sampling/importance_sampling_ratio/mean": 1.0052249431610107, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3980422429740429, "clip_ratio/low_mean": 0.040312367025762796, "clip_ratio/low_min": 0.040312367025762796, "clip_ratio/high_mean": 0.02632575877942145, "clip_ratio/high_max": 0.02632575877942145, "clip_ratio/region_mean": 0.06663812580518425, "reward_total_mean": 0.683832049369812, "reward_meter_mean": 0.6561235189437866, "reward_meter_std": 0.25921109318733215, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9382648468017578, "reward_repeat_soft_std": 0.03850472345948219, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.683832049369812, "reward_total_composite_std": 0.14870306849479675} {"timestamp_utc": "2026-04-13T06:16:59Z", "mode": "train", "global_step": 3169, "epoch": 0.3183324962330487, "loss": 0.0095, "grad_norm": 19.194238662719727, "learning_rate": 4.0000000000000003e-07, "num_tokens": 5945295.0, "completions/mean_length": 30.375, "completions/min_length": 25.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.375, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9532397985458374, "rewards/meter/std": 0.09373138099908829, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8079578876495361, "rewards/total_composite/std": 0.04142478480935097, "reward": 0.8079578876495361, "reward_std": 0.041424788534641266, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07160726934671402, "sampling/sampling_logp_difference/max": 0.888697624206543, "sampling/importance_sampling_ratio/min": 0.4111909568309784, "sampling/importance_sampling_ratio/mean": 1.0096477270126343, "sampling/importance_sampling_ratio/max": 1.6892673969268799, "entropy": 0.4385637901723385, "clip_ratio/low_mean": 0.008333333767950535, "clip_ratio/low_min": 0.008333333767950535, "clip_ratio/high_mean": 0.07281241845339537, "clip_ratio/high_max": 0.07281241845339537, "clip_ratio/region_mean": 0.0811457522213459, "reward_total_mean": 0.8079578876495361, "reward_meter_mean": 0.9532397985458374, "reward_meter_std": 0.09373138099908829, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8079578876495361, "reward_total_composite_std": 0.04142478480935097} {"timestamp_utc": "2026-04-13T06:17:07Z", "mode": "train", "global_step": 3170, "epoch": 0.3184329482672024, "loss": 0.026, "grad_norm": 6.802310943603516, "learning_rate": 3.9696969696969697e-07, "num_tokens": 5947978.0, "completions/mean_length": 139.375, "completions/min_length": 134.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 139.375, "completions/min_terminated_length": 134.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.9867603182792664, "rewards/meter/std": 0.0018439235864207149, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7949120998382568, "rewards/repeat_soft/std": 0.09796571731567383, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.8119083642959595, "rewards/total_composite/std": 0.05365743860602379, "reward": 0.8119083642959595, "reward_std": 0.05365743488073349, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06680772453546524, "sampling/sampling_logp_difference/max": 1.819101333618164, "sampling/importance_sampling_ratio/min": 0.16217142343521118, "sampling/importance_sampling_ratio/mean": 1.0058355331420898, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3069674149155617, "clip_ratio/low_mean": 0.05483822105452418, "clip_ratio/low_min": 0.05483822105452418, "clip_ratio/high_mean": 0.007352941203862429, "clip_ratio/high_max": 0.007352941203862429, "clip_ratio/region_mean": 0.06219116225838661, "reward_total_mean": 0.8119083642959595, "reward_meter_mean": 0.9867603182792664, "reward_meter_std": 0.0018439235864207149, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7949120998382568, "reward_repeat_soft_std": 0.09796571731567383, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.8119083642959595, "reward_total_composite_std": 0.05365743860602379} {"timestamp_utc": "2026-04-13T06:17:13Z", "mode": "train", "global_step": 3171, "epoch": 0.31853340030135613, "loss": 0.0699, "grad_norm": 12.509806632995605, "learning_rate": 3.9393939393939396e-07, "num_tokens": 5949407.0, "completions/mean_length": 24.625, "completions/min_length": 23.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.625, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.9872546792030334, "rewards/meter/std": 0.0014813675079494715, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8221396207809448, "rewards/total_composite/std": 0.004810743033885956, "reward": 0.8221396207809448, "reward_std": 0.004810738377273083, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11019028723239899, "sampling/sampling_logp_difference/max": 3.1550512313842773, "sampling/importance_sampling_ratio/min": 0.042636219412088394, "sampling/importance_sampling_ratio/mean": 1.0108983516693115, "sampling/importance_sampling_ratio/max": 1.9642211198806763, "entropy": 0.450157318264246, "clip_ratio/low_mean": 0.044159545097500086, "clip_ratio/low_min": 0.044159545097500086, "clip_ratio/high_mean": 0.051684783305972815, "clip_ratio/high_max": 0.051684783305972815, "clip_ratio/region_mean": 0.0958443284034729, "reward_total_mean": 0.8221396207809448, "reward_meter_mean": 0.9872546792030334, "reward_meter_std": 0.0014813675079494715, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8221396207809448, "reward_total_composite_std": 0.004810743033885956} {"timestamp_utc": "2026-04-13T06:17:20Z", "mode": "train", "global_step": 3172, "epoch": 0.3186338523355098, "loss": 0.0782, "grad_norm": 14.57606029510498, "learning_rate": 3.9090909090909095e-07, "num_tokens": 5950963.0, "completions/mean_length": 40.5, "completions/min_length": 37.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.5, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.6834734678268433, "rewards/meter/std": 0.296627014875412, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9784492254257202, "rewards/repeat_soft/std": 0.04214822128415108, "rewards/judge_quality/mean": 0.6112499833106995, "rewards/judge_quality/std": 0.25587037205696106, "rewards/total_composite/mean": 0.7387830018997192, "rewards/total_composite/std": 0.12538619339466095, "reward": 0.7387830018997192, "reward_std": 0.12538619339466095, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10829075425863266, "sampling/sampling_logp_difference/max": 1.1289420127868652, "sampling/importance_sampling_ratio/min": 0.3529711663722992, "sampling/importance_sampling_ratio/mean": 1.0019173622131348, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37291672080755234, "clip_ratio/low_mean": 0.03260059398598969, "clip_ratio/low_min": 0.03260059398598969, "clip_ratio/high_mean": 0.07027027104049921, "clip_ratio/high_max": 0.07027027104049921, "clip_ratio/region_mean": 0.1028708650264889, "reward_total_mean": 0.7387830018997192, "reward_meter_mean": 0.6834734678268433, "reward_meter_std": 0.296627014875412, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9784492254257202, "reward_repeat_soft_std": 0.04214822128415108, "reward_judge_quality_mean": 0.6112499833106995, "reward_judge_quality_std": 0.25587037205696106, "reward_total_composite_mean": 0.7387830018997192, "reward_total_composite_std": 0.12538619339466095} {"timestamp_utc": "2026-04-13T06:17:26Z", "mode": "train", "global_step": 3173, "epoch": 0.3187343043696635, "loss": 0.0774, "grad_norm": 23.433483123779297, "learning_rate": 3.878787878787879e-07, "num_tokens": 5952746.0, "completions/mean_length": 51.875, "completions/min_length": 45.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.875, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9874077439308167, "rewards/meter/std": 0.004802023060619831, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.988697350025177, "rewards/repeat_soft/std": 0.01175928208976984, "rewards/judge_quality/mean": 0.6262500286102295, "rewards/judge_quality/std": 0.2432481348514557, "rewards/total_composite/mean": 0.8810782432556152, "rewards/total_composite/std": 0.0715264230966568, "reward": 0.8810782432556152, "reward_std": 0.0715264156460762, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10143129527568817, "sampling/sampling_logp_difference/max": 1.6384177207946777, "sampling/importance_sampling_ratio/min": 0.19428721070289612, "sampling/importance_sampling_ratio/mean": 0.9930005669593811, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4264741353690624, "clip_ratio/low_mean": 0.05823530675843358, "clip_ratio/low_min": 0.05823530675843358, "clip_ratio/high_mean": 0.039866140112280846, "clip_ratio/high_max": 0.039866140112280846, "clip_ratio/region_mean": 0.09810144687071443, "reward_total_mean": 0.8810782432556152, "reward_meter_mean": 0.9874077439308167, "reward_meter_std": 0.004802023060619831, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.988697350025177, "reward_repeat_soft_std": 0.01175928208976984, "reward_judge_quality_mean": 0.6262500286102295, "reward_judge_quality_std": 0.2432481348514557, "reward_total_composite_mean": 0.8810782432556152, "reward_total_composite_std": 0.0715264230966568} {"timestamp_utc": "2026-04-13T06:17:32Z", "mode": "train", "global_step": 3174, "epoch": 0.3188347564038172, "loss": 0.0503, "grad_norm": 9.063715934753418, "learning_rate": 3.848484848484849e-07, "num_tokens": 5954421.0, "completions/mean_length": 56.375, "completions/min_length": 51.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.375, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9600424766540527, "rewards/meter/std": 0.04929869621992111, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9675877094268799, "rewards/repeat_soft/std": 0.02723308466374874, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.8092778921127319, "rewards/total_composite/std": 0.022849099710583687, "reward": 0.8092778921127319, "reward_std": 0.02284909412264824, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10969283431768417, "sampling/sampling_logp_difference/max": 2.393259048461914, "sampling/importance_sampling_ratio/min": 0.09133154153823853, "sampling/importance_sampling_ratio/mean": 1.001267671585083, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4195518083870411, "clip_ratio/low_mean": 0.042411877773702145, "clip_ratio/low_min": 0.042411877773702145, "clip_ratio/high_mean": 0.05340567883104086, "clip_ratio/high_max": 0.05340567883104086, "clip_ratio/region_mean": 0.095817556604743, "reward_total_mean": 0.8092778921127319, "reward_meter_mean": 0.9600424766540527, "reward_meter_std": 0.04929869621992111, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9675877094268799, "reward_repeat_soft_std": 0.02723308466374874, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.8092778921127319, "reward_total_composite_std": 0.022849099710583687} {"timestamp_utc": "2026-04-13T06:17:44Z", "mode": "train", "global_step": 3175, "epoch": 0.31893520843797085, "loss": -0.0918, "grad_norm": 2.2964680194854736, "learning_rate": 3.8181818181818187e-07, "num_tokens": 5955852.0, "completions/mean_length": 90.875, "completions/min_length": 29.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 30.71428680419922, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.8824761509895325, "rewards/meter/std": 0.27920129895210266, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9585226774215698, "rewards/repeat_soft/std": 0.011249415576457977, "rewards/judge_quality/mean": 0.39625000953674316, "rewards/judge_quality/std": 0.14029940962791443, "rewards/total_composite/mean": 0.6784638166427612, "rewards/total_composite/std": 0.30180126428604126, "reward": 0.6784638166427612, "reward_std": 0.30180126428604126, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0875488743185997, "sampling/sampling_logp_difference/max": 1.8897380828857422, "sampling/importance_sampling_ratio/min": 0.15111137926578522, "sampling/importance_sampling_ratio/mean": 1.0099684000015259, "sampling/importance_sampling_ratio/max": 1.7977670431137085, "entropy": 0.38145510107278824, "clip_ratio/low_mean": 0.02916666679084301, "clip_ratio/low_min": 0.02916666679084301, "clip_ratio/high_mean": 0.06184941902756691, "clip_ratio/high_max": 0.06184941902756691, "clip_ratio/region_mean": 0.09101608581840992, "reward_total_mean": 0.6784638166427612, "reward_meter_mean": 0.8824761509895325, "reward_meter_std": 0.27920129895210266, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9585226774215698, "reward_repeat_soft_std": 0.011249415576457977, "reward_judge_quality_mean": 0.39625000953674316, "reward_judge_quality_std": 0.14029940962791443, "reward_total_composite_mean": 0.6784638166427612, "reward_total_composite_std": 0.30180126428604126} {"timestamp_utc": "2026-04-13T06:17:50Z", "mode": "train", "global_step": 3176, "epoch": 0.31903566047212456, "loss": -0.0351, "grad_norm": 10.81485366821289, "learning_rate": 3.787878787878788e-07, "num_tokens": 5957535.0, "completions/mean_length": 54.375, "completions/min_length": 47.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.375, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.7787264585494995, "rewards/meter/std": 0.36807680130004883, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9849765300750732, "rewards/repeat_soft/std": 0.014601239934563637, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.25587037205696106, "rewards/total_composite/mean": 0.7822996377944946, "rewards/total_composite/std": 0.16523513197898865, "reward": 0.7822996377944946, "reward_std": 0.16523513197898865, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11334329843521118, "sampling/sampling_logp_difference/max": 1.921271562576294, "sampling/importance_sampling_ratio/min": 0.1464206576347351, "sampling/importance_sampling_ratio/mean": 1.006205677986145, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6014282032847404, "clip_ratio/low_mean": 0.019742225762456656, "clip_ratio/low_min": 0.019742225762456656, "clip_ratio/high_mean": 0.08110572397708893, "clip_ratio/high_max": 0.08110572397708893, "clip_ratio/region_mean": 0.10084794973954558, "reward_total_mean": 0.7822996377944946, "reward_meter_mean": 0.7787264585494995, "reward_meter_std": 0.36807680130004883, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9849765300750732, "reward_repeat_soft_std": 0.014601239934563637, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.25587037205696106, "reward_total_composite_mean": 0.7822996377944946, "reward_total_composite_std": 0.16523513197898865} {"timestamp_utc": "2026-04-13T06:17:56Z", "mode": "train", "global_step": 3177, "epoch": 0.31913611250627827, "loss": -0.0025, "grad_norm": 8.869209289550781, "learning_rate": 3.757575757575758e-07, "num_tokens": 5959170.0, "completions/mean_length": 54.375, "completions/min_length": 48.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.375, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9437544941902161, "rewards/meter/std": 0.07108517736196518, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9824265241622925, "rewards/repeat_soft/std": 0.010148690082132816, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.8034321665763855, "rewards/total_composite/std": 0.033331844955682755, "reward": 0.8034321665763855, "reward_std": 0.03333185240626335, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09370190650224686, "sampling/sampling_logp_difference/max": 1.8541202545166016, "sampling/importance_sampling_ratio/min": 0.15659065544605255, "sampling/importance_sampling_ratio/mean": 0.9967269897460938, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4678680747747421, "clip_ratio/low_mean": 0.027960526756942272, "clip_ratio/low_min": 0.027960526756942272, "clip_ratio/high_mean": 0.07651232462376356, "clip_ratio/high_max": 0.07651232462376356, "clip_ratio/region_mean": 0.10447285138070583, "reward_total_mean": 0.8034321665763855, "reward_meter_mean": 0.9437544941902161, "reward_meter_std": 0.07108517736196518, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9824265241622925, "reward_repeat_soft_std": 0.010148690082132816, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.8034321665763855, "reward_total_composite_std": 0.033331844955682755} {"timestamp_utc": "2026-04-13T06:18:03Z", "mode": "train", "global_step": 3178, "epoch": 0.3192365645404319, "loss": 0.0393, "grad_norm": 9.231740951538086, "learning_rate": 3.7272727272727274e-07, "num_tokens": 5960669.0, "completions/mean_length": 28.375, "completions/min_length": 24.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.375, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9865385293960571, "rewards/meter/std": 0.009110051207244396, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9252045154571533, "rewards/repeat_soft/std": 0.05422128736972809, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.8312128186225891, "rewards/total_composite/std": 0.04868081584572792, "reward": 0.8312128186225891, "reward_std": 0.04868079721927643, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08517567813396454, "sampling/sampling_logp_difference/max": 1.413257360458374, "sampling/importance_sampling_ratio/min": 0.24334931373596191, "sampling/importance_sampling_ratio/mean": 1.005692720413208, "sampling/importance_sampling_ratio/max": 1.63442063331604, "entropy": 0.39530643075704575, "clip_ratio/low_mean": 0.10834712255746126, "clip_ratio/low_min": 0.10834712255746126, "clip_ratio/high_mean": 0.004629629664123058, "clip_ratio/high_max": 0.004629629664123058, "clip_ratio/region_mean": 0.11297675222158432, "reward_total_mean": 0.8312128186225891, "reward_meter_mean": 0.9865385293960571, "reward_meter_std": 0.009110051207244396, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9252045154571533, "reward_repeat_soft_std": 0.05422128736972809, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.8312128186225891, "reward_total_composite_std": 0.04868081584572792} {"timestamp_utc": "2026-04-13T06:18:09Z", "mode": "train", "global_step": 3179, "epoch": 0.31933701657458563, "loss": -0.0228, "grad_norm": 11.899554252624512, "learning_rate": 3.6969696969696973e-07, "num_tokens": 5962279.0, "completions/mean_length": 55.25, "completions/min_length": 48.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.25, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.8573475480079651, "rewards/meter/std": 0.24903464317321777, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9943018555641174, "rewards/repeat_soft/std": 0.007124005351215601, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.7811115980148315, "rewards/total_composite/std": 0.12639622390270233, "reward": 0.7811115980148315, "reward_std": 0.12639620900154114, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11034230887889862, "sampling/sampling_logp_difference/max": 2.0037050247192383, "sampling/importance_sampling_ratio/min": 0.13483478128910065, "sampling/importance_sampling_ratio/mean": 0.9987104535102844, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.431323766708374, "clip_ratio/low_mean": 0.015471813501790166, "clip_ratio/low_min": 0.015471813501790166, "clip_ratio/high_mean": 0.0689552326221019, "clip_ratio/high_max": 0.0689552326221019, "clip_ratio/region_mean": 0.08442704612389207, "reward_total_mean": 0.7811115980148315, "reward_meter_mean": 0.8573475480079651, "reward_meter_std": 0.24903464317321777, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9943018555641174, "reward_repeat_soft_std": 0.007124005351215601, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.7811115980148315, "reward_total_composite_std": 0.12639622390270233} {"timestamp_utc": "2026-04-13T06:18:16Z", "mode": "train", "global_step": 3180, "epoch": 0.31943746860873934, "loss": -0.0026, "grad_norm": 12.025635719299316, "learning_rate": 3.666666666666667e-07, "num_tokens": 5964268.0, "completions/mean_length": 86.625, "completions/min_length": 75.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.625, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.8846205472946167, "rewards/meter/std": 0.2858838140964508, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9630615711212158, "rewards/repeat_soft/std": 0.0228089801967144, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7703854441642761, "rewards/total_composite/std": 0.1284448206424713, "reward": 0.7703854441642761, "reward_std": 0.12844480574131012, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10333739966154099, "sampling/sampling_logp_difference/max": 2.2183899879455566, "sampling/importance_sampling_ratio/min": 0.10878410935401917, "sampling/importance_sampling_ratio/mean": 0.9962204098701477, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4265640079975128, "clip_ratio/low_mean": 0.0062500000931322575, "clip_ratio/low_min": 0.0062500000931322575, "clip_ratio/high_mean": 0.08403525035828352, "clip_ratio/high_max": 0.08403525035828352, "clip_ratio/region_mean": 0.09028525045141578, "reward_total_mean": 0.7703854441642761, "reward_meter_mean": 0.8846205472946167, "reward_meter_std": 0.2858838140964508, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9630615711212158, "reward_repeat_soft_std": 0.0228089801967144, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7703854441642761, "reward_total_composite_std": 0.1284448206424713} {"timestamp_utc": "2026-04-13T06:18:24Z", "mode": "train", "global_step": 3181, "epoch": 0.319537920642893, "loss": 0.0125, "grad_norm": 5.000176429748535, "learning_rate": 3.6363636363636366e-07, "num_tokens": 5967239.0, "completions/mean_length": 167.375, "completions/min_length": 136.0, "completions/max_length": 184.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 167.375, "completions/min_terminated_length": 136.0, "completions/max_terminated_length": 184.0, "rewards/meter/mean": 0.9069427847862244, "rewards/meter/std": 0.22546373307704926, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8986557722091675, "rewards/repeat_soft/std": 0.051688775420188904, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7627397775650024, "rewards/total_composite/std": 0.09588442742824554, "reward": 0.7627397775650024, "reward_std": 0.09588442742824554, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08588375896215439, "sampling/sampling_logp_difference/max": 4.2809038162231445, "sampling/importance_sampling_ratio/min": 0.013830156065523624, "sampling/importance_sampling_ratio/mean": 1.0018832683563232, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3727256618440151, "clip_ratio/low_mean": 0.010294117964804173, "clip_ratio/low_min": 0.010294117964804173, "clip_ratio/high_mean": 0.07279908796772361, "clip_ratio/high_max": 0.07279908796772361, "clip_ratio/region_mean": 0.08309320593252778, "reward_total_mean": 0.7627397775650024, "reward_meter_mean": 0.9069427847862244, "reward_meter_std": 0.22546373307704926, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8986557722091675, "reward_repeat_soft_std": 0.051688775420188904, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7627397775650024, "reward_total_composite_std": 0.09588442742824554} {"timestamp_utc": "2026-04-13T06:18:30Z", "mode": "train", "global_step": 3182, "epoch": 0.3196383726770467, "loss": 0.0293, "grad_norm": 8.995685577392578, "learning_rate": 3.6060606060606065e-07, "num_tokens": 5968994.0, "completions/mean_length": 51.375, "completions/min_length": 46.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.375, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9520926475524902, "rewards/meter/std": 0.09461932629346848, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9937039613723755, "rewards/repeat_soft/std": 0.007737454958260059, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.8424370884895325, "rewards/total_composite/std": 0.08962354063987732, "reward": 0.8424370884895325, "reward_std": 0.08962354809045792, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06825825572013855, "sampling/sampling_logp_difference/max": 0.8660565614700317, "sampling/importance_sampling_ratio/min": 0.42060691118240356, "sampling/importance_sampling_ratio/mean": 1.0128819942474365, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.343615535646677, "clip_ratio/low_mean": 0.05653283605352044, "clip_ratio/low_min": 0.05653283605352044, "clip_ratio/high_mean": 0.009928929852321744, "clip_ratio/high_max": 0.009928929852321744, "clip_ratio/region_mean": 0.06646176590584219, "reward_total_mean": 0.8424370884895325, "reward_meter_mean": 0.9520926475524902, "reward_meter_std": 0.09461932629346848, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9937039613723755, "reward_repeat_soft_std": 0.007737454958260059, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.8424370884895325, "reward_total_composite_std": 0.08962354063987732} {"timestamp_utc": "2026-04-13T06:18:38Z", "mode": "train", "global_step": 3183, "epoch": 0.3197388247112004, "loss": 0.0064, "grad_norm": 4.781403541564941, "learning_rate": 3.5757575757575764e-07, "num_tokens": 5971526.0, "completions/mean_length": 140.5, "completions/min_length": 131.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 140.5, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.9889276027679443, "rewards/meter/std": 0.00405951589345932, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8269180655479431, "rewards/repeat_soft/std": 0.06444202363491058, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8037092089653015, "rewards/total_composite/std": 0.006718754302710295, "reward": 0.8037092089653015, "reward_std": 0.006718754302710295, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07416502386331558, "sampling/sampling_logp_difference/max": 2.3082785606384277, "sampling/importance_sampling_ratio/min": 0.09943227469921112, "sampling/importance_sampling_ratio/mean": 0.9949133396148682, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31679889373481274, "clip_ratio/low_mean": 0.018465064465999603, "clip_ratio/low_min": 0.018465064465999603, "clip_ratio/high_mean": 0.05400266079232097, "clip_ratio/high_max": 0.05400266079232097, "clip_ratio/region_mean": 0.07246772525832057, "reward_total_mean": 0.8037092089653015, "reward_meter_mean": 0.9889276027679443, "reward_meter_std": 0.00405951589345932, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8269180655479431, "reward_repeat_soft_std": 0.06444202363491058, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8037092089653015, "reward_total_composite_std": 0.006718754302710295} {"timestamp_utc": "2026-04-13T06:18:45Z", "mode": "train", "global_step": 3184, "epoch": 0.3198392767453541, "loss": 0.0015, "grad_norm": 7.164723873138428, "learning_rate": 3.545454545454546e-07, "num_tokens": 5973306.0, "completions/mean_length": 68.5, "completions/min_length": 64.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.5, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9895360469818115, "rewards/meter/std": 0.0037693430203944445, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9822664260864258, "rewards/repeat_soft/std": 0.03236118331551552, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.1865811049938202, "rewards/total_composite/mean": 0.8528928756713867, "rewards/total_composite/std": 0.056238047778606415, "reward": 0.8528928756713867, "reward_std": 0.056238044053316116, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10704988986253738, "sampling/sampling_logp_difference/max": 2.292570114135742, "sampling/importance_sampling_ratio/min": 0.10100653022527695, "sampling/importance_sampling_ratio/mean": 0.998375415802002, "sampling/importance_sampling_ratio/max": 1.838136911392212, "entropy": 0.4067746624350548, "clip_ratio/low_mean": 0.07504863664507866, "clip_ratio/low_min": 0.07504863664507866, "clip_ratio/high_mean": 0.03368293680250645, "clip_ratio/high_max": 0.03368293680250645, "clip_ratio/region_mean": 0.1087315734475851, "reward_total_mean": 0.8528928756713867, "reward_meter_mean": 0.9895360469818115, "reward_meter_std": 0.0037693430203944445, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9822664260864258, "reward_repeat_soft_std": 0.03236118331551552, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.1865811049938202, "reward_total_composite_mean": 0.8528928756713867, "reward_total_composite_std": 0.056238047778606415} {"timestamp_utc": "2026-04-13T06:18:52Z", "mode": "train", "global_step": 3185, "epoch": 0.31993972877950777, "loss": 0.0446, "grad_norm": 5.852051734924316, "learning_rate": 3.515151515151515e-07, "num_tokens": 5975547.0, "completions/mean_length": 116.125, "completions/min_length": 102.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.125, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.9719011187553406, "rewards/meter/std": 0.007002949248999357, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7437776327133179, "rewards/repeat_soft/std": 0.06793434917926788, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7738583087921143, "rewards/total_composite/std": 0.01898610033094883, "reward": 0.7738583087921143, "reward_std": 0.01898610033094883, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.048396944999694824, "sampling/sampling_logp_difference/max": 1.513561725616455, "sampling/importance_sampling_ratio/min": 0.22012455761432648, "sampling/importance_sampling_ratio/mean": 1.002341866493225, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23592505976557732, "clip_ratio/low_mean": 0.016151844058185816, "clip_ratio/low_min": 0.016151844058185816, "clip_ratio/high_mean": 0.02797444467432797, "clip_ratio/high_max": 0.02797444467432797, "clip_ratio/region_mean": 0.044126288732513785, "reward_total_mean": 0.7738583087921143, "reward_meter_mean": 0.9719011187553406, "reward_meter_std": 0.007002949248999357, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7437776327133179, "reward_repeat_soft_std": 0.06793434917926788, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7738583087921143, "reward_total_composite_std": 0.01898610033094883} {"timestamp_utc": "2026-04-13T06:18:59Z", "mode": "train", "global_step": 3186, "epoch": 0.3200401808136615, "loss": 0.0045, "grad_norm": 8.88284969329834, "learning_rate": 3.4848484848484856e-07, "num_tokens": 5977374.0, "completions/mean_length": 65.375, "completions/min_length": 61.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.375, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.992681622505188, "rewards/meter/std": 0.004721727222204208, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9601198434829712, "rewards/repeat_soft/std": 0.036284174770116806, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.10392305999994278, "rewards/total_composite/mean": 0.8322187066078186, "rewards/total_composite/std": 0.03320806473493576, "reward": 0.8322187066078186, "reward_std": 0.033208079636096954, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08885729312896729, "sampling/sampling_logp_difference/max": 1.2231876850128174, "sampling/importance_sampling_ratio/min": 0.29429057240486145, "sampling/importance_sampling_ratio/mean": 1.0056885480880737, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4857178069651127, "clip_ratio/low_mean": 0.08388576470315456, "clip_ratio/low_min": 0.08388576470315456, "clip_ratio/high_mean": 0.014705882407724857, "clip_ratio/high_max": 0.014705882407724857, "clip_ratio/region_mean": 0.09859164711087942, "reward_total_mean": 0.8322187066078186, "reward_meter_mean": 0.992681622505188, "reward_meter_std": 0.004721727222204208, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9601198434829712, "reward_repeat_soft_std": 0.036284174770116806, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.10392305999994278, "reward_total_composite_mean": 0.8322187066078186, "reward_total_composite_std": 0.03320806473493576} {"timestamp_utc": "2026-04-13T06:19:07Z", "mode": "train", "global_step": 3187, "epoch": 0.3201406328478152, "loss": 0.0137, "grad_norm": 6.328986167907715, "learning_rate": 3.454545454545455e-07, "num_tokens": 5980498.0, "completions/mean_length": 172.5, "completions/min_length": 153.0, "completions/max_length": 181.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 172.5, "completions/min_terminated_length": 153.0, "completions/max_terminated_length": 181.0, "rewards/meter/mean": 0.9891986846923828, "rewards/meter/std": 0.005989863537251949, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7530725598335266, "rewards/repeat_soft/std": 0.051139622926712036, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7836966514587402, "rewards/total_composite/std": 0.02320719137787819, "reward": 0.7836966514587402, "reward_std": 0.02320718951523304, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08278960734605789, "sampling/sampling_logp_difference/max": 2.3793206214904785, "sampling/importance_sampling_ratio/min": 0.09261347353458405, "sampling/importance_sampling_ratio/mean": 0.9935018420219421, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3394974581897259, "clip_ratio/low_mean": 0.02085035853087902, "clip_ratio/low_min": 0.02085035853087902, "clip_ratio/high_mean": 0.039229005109518766, "clip_ratio/high_max": 0.039229005109518766, "clip_ratio/region_mean": 0.06007936364039779, "reward_total_mean": 0.7836966514587402, "reward_meter_mean": 0.9891986846923828, "reward_meter_std": 0.005989863537251949, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7530725598335266, "reward_repeat_soft_std": 0.051139622926712036, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7836966514587402, "reward_total_composite_std": 0.02320719137787819} {"timestamp_utc": "2026-04-13T06:19:13Z", "mode": "train", "global_step": 3188, "epoch": 0.32024108488196884, "loss": -0.0173, "grad_norm": 12.216792106628418, "learning_rate": 3.4242424242424243e-07, "num_tokens": 5981941.0, "completions/mean_length": 26.375, "completions/min_length": 22.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.375, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.9928655624389648, "rewards/meter/std": 0.004633770324289799, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9604166746139526, "rewards/repeat_soft/std": 0.00589255103841424, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8199561834335327, "rewards/total_composite/std": 0.004281556699424982, "reward": 0.8199561834335327, "reward_std": 0.004281552042812109, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09490005671977997, "sampling/sampling_logp_difference/max": 1.9640140533447266, "sampling/importance_sampling_ratio/min": 0.1402941346168518, "sampling/importance_sampling_ratio/mean": 1.0007041692733765, "sampling/importance_sampling_ratio/max": 1.8680721521377563, "entropy": 0.408481203019619, "clip_ratio/low_mean": 0.06600108370184898, "clip_ratio/low_min": 0.06600108370184898, "clip_ratio/high_mean": 0.04408487165346742, "clip_ratio/high_max": 0.04408487165346742, "clip_ratio/region_mean": 0.1100859553553164, "reward_total_mean": 0.8199561834335327, "reward_meter_mean": 0.9928655624389648, "reward_meter_std": 0.004633770324289799, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9604166746139526, "reward_repeat_soft_std": 0.00589255103841424, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8199561834335327, "reward_total_composite_std": 0.004281556699424982} {"timestamp_utc": "2026-04-13T06:19:19Z", "mode": "train", "global_step": 3189, "epoch": 0.32034153691612255, "loss": 0.0129, "grad_norm": 9.311483383178711, "learning_rate": 3.393939393939395e-07, "num_tokens": 5983681.0, "completions/mean_length": 56.5, "completions/min_length": 51.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.5, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.5384381413459778, "rewards/meter/std": 0.3147052228450775, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8888794183731079, "rewards/repeat_soft/std": 0.02458801120519638, "rewards/judge_quality/mean": 0.42124998569488525, "rewards/judge_quality/std": 0.06998724490404129, "rewards/total_composite/mean": 0.6075600981712341, "rewards/total_composite/std": 0.12923958897590637, "reward": 0.6075600981712341, "reward_std": 0.12923958897590637, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06805529445409775, "sampling/sampling_logp_difference/max": 1.7802863121032715, "sampling/importance_sampling_ratio/min": 0.16858987510204315, "sampling/importance_sampling_ratio/mean": 1.011253833770752, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.32275932282209396, "clip_ratio/low_mean": 0.04009130736812949, "clip_ratio/low_min": 0.04009130736812949, "clip_ratio/high_mean": 0.01085043977946043, "clip_ratio/high_max": 0.01085043977946043, "clip_ratio/region_mean": 0.05094174714758992, "reward_total_mean": 0.6075600981712341, "reward_meter_mean": 0.5384381413459778, "reward_meter_std": 0.3147052228450775, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8888794183731079, "reward_repeat_soft_std": 0.02458801120519638, "reward_judge_quality_mean": 0.42124998569488525, "reward_judge_quality_std": 0.06998724490404129, "reward_total_composite_mean": 0.6075600981712341, "reward_total_composite_std": 0.12923958897590637} {"timestamp_utc": "2026-04-13T06:19:26Z", "mode": "train", "global_step": 3190, "epoch": 0.32044198895027626, "loss": 0.0173, "grad_norm": 8.603158950805664, "learning_rate": 3.363636363636364e-07, "num_tokens": 5985421.0, "completions/mean_length": 69.5, "completions/min_length": 62.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.5, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9955778121948242, "rewards/meter/std": 0.0021531383972615004, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9649364352226257, "rewards/repeat_soft/std": 0.02948417328298092, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.8328787088394165, "rewards/total_composite/std": 0.05888361111283302, "reward": 0.8328787088394165, "reward_std": 0.05888361856341362, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09075454622507095, "sampling/sampling_logp_difference/max": 1.133615493774414, "sampling/importance_sampling_ratio/min": 0.3218674659729004, "sampling/importance_sampling_ratio/mean": 0.999968409538269, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45678364858031273, "clip_ratio/low_mean": 0.058139098808169365, "clip_ratio/low_min": 0.058139098808169365, "clip_ratio/high_mean": 0.017605634406208992, "clip_ratio/high_max": 0.017605634406208992, "clip_ratio/region_mean": 0.07574473321437836, "reward_total_mean": 0.8328787088394165, "reward_meter_mean": 0.9955778121948242, "reward_meter_std": 0.0021531383972615004, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9649364352226257, "reward_repeat_soft_std": 0.02948417328298092, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.8328787088394165, "reward_total_composite_std": 0.05888361111283302} {"timestamp_utc": "2026-04-13T06:19:32Z", "mode": "train", "global_step": 3191, "epoch": 0.3205424409844299, "loss": -0.0182, "grad_norm": 13.186880111694336, "learning_rate": 3.3333333333333335e-07, "num_tokens": 5986854.0, "completions/mean_length": 28.125, "completions/min_length": 25.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.125, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9849893450737, "rewards/meter/std": 0.004496193025261164, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.3499999940395355, "rewards/judge_quality/std": 0.10690449178218842, "rewards/total_composite/mean": 0.7944952249526978, "rewards/total_composite/std": 0.03325314074754715, "reward": 0.7944952249526978, "reward_std": 0.03325314447283745, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10614202171564102, "sampling/sampling_logp_difference/max": 3.1641592979431152, "sampling/importance_sampling_ratio/min": 0.042249646037817, "sampling/importance_sampling_ratio/mean": 0.9925504922866821, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4748496524989605, "clip_ratio/low_mean": 0.02358974376693368, "clip_ratio/low_min": 0.02358974376693368, "clip_ratio/high_mean": 0.05955905839800835, "clip_ratio/high_max": 0.05955905839800835, "clip_ratio/region_mean": 0.08314880216494203, "reward_total_mean": 0.7944952249526978, "reward_meter_mean": 0.9849893450737, "reward_meter_std": 0.004496193025261164, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.3499999940395355, "reward_judge_quality_std": 0.10690449178218842, "reward_total_composite_mean": 0.7944952249526978, "reward_total_composite_std": 0.03325314074754715} {"timestamp_utc": "2026-04-13T06:19:38Z", "mode": "train", "global_step": 3192, "epoch": 0.3206428930185836, "loss": -0.0163, "grad_norm": 11.26955795288086, "learning_rate": 3.303030303030303e-07, "num_tokens": 5988503.0, "completions/mean_length": 61.125, "completions/min_length": 54.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.125, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9856968522071838, "rewards/meter/std": 0.009364540688693523, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.948104739189148, "rewards/repeat_soft/std": 0.058360371738672256, "rewards/judge_quality/mean": 0.8362500667572021, "rewards/judge_quality/std": 0.1784006506204605, "rewards/total_composite/mean": 0.9392490983009338, "rewards/total_composite/std": 0.05821295827627182, "reward": 0.9392490983009338, "reward_std": 0.05821295082569122, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07777315378189087, "sampling/sampling_logp_difference/max": 1.3560599088668823, "sampling/importance_sampling_ratio/min": 0.25767403841018677, "sampling/importance_sampling_ratio/mean": 0.9944658875465393, "sampling/importance_sampling_ratio/max": 1.7576546669006348, "entropy": 0.3769986145198345, "clip_ratio/low_mean": 0.012962963432073593, "clip_ratio/low_min": 0.012962963432073593, "clip_ratio/high_mean": 0.06432176288217306, "clip_ratio/high_max": 0.06432176288217306, "clip_ratio/region_mean": 0.07728472631424665, "reward_total_mean": 0.9392490983009338, "reward_meter_mean": 0.9856968522071838, "reward_meter_std": 0.009364540688693523, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.948104739189148, "reward_repeat_soft_std": 0.058360371738672256, "reward_judge_quality_mean": 0.8362500667572021, "reward_judge_quality_std": 0.1784006506204605, "reward_total_composite_mean": 0.9392490983009338, "reward_total_composite_std": 0.05821295827627182} {"timestamp_utc": "2026-04-13T06:19:44Z", "mode": "train", "global_step": 3193, "epoch": 0.3207433450527373, "loss": 0.0245, "grad_norm": 6.176050662994385, "learning_rate": 3.2727272727272733e-07, "num_tokens": 5990060.0, "completions/mean_length": 45.625, "completions/min_length": 44.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.625, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.9657387733459473, "rewards/meter/std": 0.00710863433778286, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8341579437255859, "rewards/repeat_soft/std": 0.09480934590101242, "rewards/judge_quality/mean": 0.3075000047683716, "rewards/judge_quality/std": 0.09467990696430206, "rewards/total_composite/mean": 0.7602482438087463, "rewards/total_composite/std": 0.030139554291963577, "reward": 0.7602482438087463, "reward_std": 0.030139556154608727, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05492093414068222, "sampling/sampling_logp_difference/max": 2.1740760803222656, "sampling/importance_sampling_ratio/min": 0.11371316760778427, "sampling/importance_sampling_ratio/mean": 1.0023781061172485, "sampling/importance_sampling_ratio/max": 1.9965397119522095, "entropy": 0.23370471596717834, "clip_ratio/low_mean": 0.01902173925191164, "clip_ratio/low_min": 0.01902173925191164, "clip_ratio/high_mean": 0.019639328587800264, "clip_ratio/high_max": 0.019639328587800264, "clip_ratio/region_mean": 0.038661067839711905, "reward_total_mean": 0.7602482438087463, "reward_meter_mean": 0.9657387733459473, "reward_meter_std": 0.00710863433778286, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8341579437255859, "reward_repeat_soft_std": 0.09480934590101242, "reward_judge_quality_mean": 0.3075000047683716, "reward_judge_quality_std": 0.09467990696430206, "reward_total_composite_mean": 0.7602482438087463, "reward_total_composite_std": 0.030139554291963577} {"timestamp_utc": "2026-04-13T06:19:51Z", "mode": "train", "global_step": 3194, "epoch": 0.32084379708689104, "loss": 0.0123, "grad_norm": 5.554563999176025, "learning_rate": 3.2424242424242427e-07, "num_tokens": 5992282.0, "completions/mean_length": 90.75, "completions/min_length": 86.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.75, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.8622914552688599, "rewards/meter/std": 0.26698872447013855, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8132039308547974, "rewards/repeat_soft/std": 0.03815414011478424, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7453515529632568, "rewards/total_composite/std": 0.11966572701931, "reward": 0.7453515529632568, "reward_std": 0.1196657195687294, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04475681856274605, "sampling/sampling_logp_difference/max": 1.48199462890625, "sampling/importance_sampling_ratio/min": 0.22718410193920135, "sampling/importance_sampling_ratio/mean": 0.9995405077934265, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.21034608408808708, "clip_ratio/low_mean": 0.004076086916029453, "clip_ratio/low_min": 0.004076086916029453, "clip_ratio/high_mean": 0.0386276007629931, "clip_ratio/high_max": 0.0386276007629931, "clip_ratio/region_mean": 0.04270368767902255, "reward_total_mean": 0.7453515529632568, "reward_meter_mean": 0.8622914552688599, "reward_meter_std": 0.26698872447013855, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8132039308547974, "reward_repeat_soft_std": 0.03815414011478424, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7453515529632568, "reward_total_composite_std": 0.11966572701931} {"timestamp_utc": "2026-04-13T06:19:57Z", "mode": "train", "global_step": 3195, "epoch": 0.3209442491210447, "loss": 0.0161, "grad_norm": 19.092355728149414, "learning_rate": 3.212121212121212e-07, "num_tokens": 5993688.0, "completions/mean_length": 26.75, "completions/min_length": 24.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.75, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9876864552497864, "rewards/meter/std": 0.005100971087813377, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9360994100570679, "rewards/repeat_soft/std": 0.06524981558322906, "rewards/judge_quality/mean": 0.3999999761581421, "rewards/judge_quality/std": 0.09258200973272324, "rewards/total_composite/mean": 0.8080688118934631, "rewards/total_composite/std": 0.02607852779328823, "reward": 0.8080688118934631, "reward_std": 0.026078524067997932, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09123432636260986, "sampling/sampling_logp_difference/max": 1.7639193534851074, "sampling/importance_sampling_ratio/min": 0.17137187719345093, "sampling/importance_sampling_ratio/mean": 0.9809799790382385, "sampling/importance_sampling_ratio/max": 1.9944734573364258, "entropy": 0.4587009474635124, "clip_ratio/low_mean": 0.020089285913854837, "clip_ratio/low_min": 0.020089285913854837, "clip_ratio/high_mean": 0.05620594974607229, "clip_ratio/high_max": 0.05620594974607229, "clip_ratio/region_mean": 0.07629523565992713, "reward_total_mean": 0.8080688118934631, "reward_meter_mean": 0.9876864552497864, "reward_meter_std": 0.005100971087813377, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9360994100570679, "reward_repeat_soft_std": 0.06524981558322906, "reward_judge_quality_mean": 0.3999999761581421, "reward_judge_quality_std": 0.09258200973272324, "reward_total_composite_mean": 0.8080688118934631, "reward_total_composite_std": 0.02607852779328823} {"timestamp_utc": "2026-04-13T06:20:05Z", "mode": "train", "global_step": 3196, "epoch": 0.3210447011551984, "loss": 0.0371, "grad_norm": 7.811895847320557, "learning_rate": 3.181818181818182e-07, "num_tokens": 5996829.0, "completions/mean_length": 178.625, "completions/min_length": 173.0, "completions/max_length": 190.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 178.625, "completions/min_terminated_length": 173.0, "completions/max_terminated_length": 190.0, "rewards/meter/mean": 0.9892479777336121, "rewards/meter/std": 0.002705504186451435, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7899119853973389, "rewards/repeat_soft/std": 0.05912560224533081, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.7900277376174927, "rewards/total_composite/std": 0.026133911684155464, "reward": 0.7900277376174927, "reward_std": 0.02613391913473606, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0546642504632473, "sampling/sampling_logp_difference/max": 1.5936174392700195, "sampling/importance_sampling_ratio/min": 0.20318925380706787, "sampling/importance_sampling_ratio/mean": 1.0014605522155762, "sampling/importance_sampling_ratio/max": 1.9052642583847046, "entropy": 0.2788252290338278, "clip_ratio/low_mean": 0.010450258385390043, "clip_ratio/low_min": 0.010450258385390043, "clip_ratio/high_mean": 0.038840282475575805, "clip_ratio/high_max": 0.038840282475575805, "clip_ratio/region_mean": 0.04929054086096585, "reward_total_mean": 0.7900277376174927, "reward_meter_mean": 0.9892479777336121, "reward_meter_std": 0.002705504186451435, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7899119853973389, "reward_repeat_soft_std": 0.05912560224533081, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.7900277376174927, "reward_total_composite_std": 0.026133911684155464} {"timestamp_utc": "2026-04-13T06:20:12Z", "mode": "train", "global_step": 3197, "epoch": 0.3211451531893521, "loss": 0.0419, "grad_norm": 13.622358322143555, "learning_rate": 3.151515151515152e-07, "num_tokens": 5998393.0, "completions/mean_length": 48.5, "completions/min_length": 37.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.5, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9779934287071228, "rewards/meter/std": 0.02160756103694439, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9648788571357727, "rewards/repeat_soft/std": 0.02818004973232746, "rewards/judge_quality/mean": 0.4762499928474426, "rewards/judge_quality/std": 0.18173274397850037, "rewards/total_composite/mean": 0.8294599056243896, "rewards/total_composite/std": 0.06302401423454285, "reward": 0.8294599056243896, "reward_std": 0.06302399933338165, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11688022315502167, "sampling/sampling_logp_difference/max": 1.95894455909729, "sampling/importance_sampling_ratio/min": 0.14100715517997742, "sampling/importance_sampling_ratio/mean": 0.9987087249755859, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5922078676521778, "clip_ratio/low_mean": 0.0783074488863349, "clip_ratio/low_min": 0.0783074488863349, "clip_ratio/high_mean": 0.033976498525589705, "clip_ratio/high_max": 0.033976498525589705, "clip_ratio/region_mean": 0.1122839474119246, "reward_total_mean": 0.8294599056243896, "reward_meter_mean": 0.9779934287071228, "reward_meter_std": 0.02160756103694439, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9648788571357727, "reward_repeat_soft_std": 0.02818004973232746, "reward_judge_quality_mean": 0.4762499928474426, "reward_judge_quality_std": 0.18173274397850037, "reward_total_composite_mean": 0.8294599056243896, "reward_total_composite_std": 0.06302401423454285} {"timestamp_utc": "2026-04-13T06:20:18Z", "mode": "train", "global_step": 3198, "epoch": 0.32124560522350576, "loss": 0.0157, "grad_norm": 8.899191856384277, "learning_rate": 3.1212121212121213e-07, "num_tokens": 6000044.0, "completions/mean_length": 64.375, "completions/min_length": 61.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.375, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9827784299850464, "rewards/meter/std": 0.013956635259091854, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.970016598701477, "rewards/repeat_soft/std": 0.02862418256700039, "rewards/judge_quality/mean": 0.668749988079071, "rewards/judge_quality/std": 0.25842589139938354, "rewards/total_composite/mean": 0.8898769617080688, "rewards/total_composite/std": 0.0743117704987526, "reward": 0.8898769617080688, "reward_std": 0.07431178539991379, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09052116423845291, "sampling/sampling_logp_difference/max": 1.5626440048217773, "sampling/importance_sampling_ratio/min": 0.209581196308136, "sampling/importance_sampling_ratio/mean": 1.0155178308486938, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.42778006568551064, "clip_ratio/low_mean": 0.04647779371589422, "clip_ratio/low_min": 0.04647779371589422, "clip_ratio/high_mean": 0.045570459216833115, "clip_ratio/high_max": 0.045570459216833115, "clip_ratio/region_mean": 0.09204825293272734, "reward_total_mean": 0.8898769617080688, "reward_meter_mean": 0.9827784299850464, "reward_meter_std": 0.013956635259091854, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.970016598701477, "reward_repeat_soft_std": 0.02862418256700039, "reward_judge_quality_mean": 0.668749988079071, "reward_judge_quality_std": 0.25842589139938354, "reward_total_composite_mean": 0.8898769617080688, "reward_total_composite_std": 0.0743117704987526} {"timestamp_utc": "2026-04-13T06:20:24Z", "mode": "train", "global_step": 3199, "epoch": 0.32134605725765947, "loss": 0.0089, "grad_norm": 8.344149589538574, "learning_rate": 3.090909090909091e-07, "num_tokens": 6001870.0, "completions/mean_length": 61.25, "completions/min_length": 59.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.25, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9863761067390442, "rewards/meter/std": 0.0035213471855968237, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9211262464523315, "rewards/repeat_soft/std": 0.05527299642562866, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.813106894493103, "rewards/total_composite/std": 0.007306722924113274, "reward": 0.813106894493103, "reward_std": 0.0073067243210971355, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.062430545687675476, "sampling/sampling_logp_difference/max": 1.2790019512176514, "sampling/importance_sampling_ratio/min": 0.2783149182796478, "sampling/importance_sampling_ratio/mean": 0.9975454807281494, "sampling/importance_sampling_ratio/max": 1.6910251379013062, "entropy": 0.2800346687436104, "clip_ratio/low_mean": 0.02033079625107348, "clip_ratio/low_min": 0.02033079625107348, "clip_ratio/high_mean": 0.037113590631634, "clip_ratio/high_max": 0.037113590631634, "clip_ratio/region_mean": 0.05744438688270748, "reward_total_mean": 0.813106894493103, "reward_meter_mean": 0.9863761067390442, "reward_meter_std": 0.0035213471855968237, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9211262464523315, "reward_repeat_soft_std": 0.05527299642562866, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.813106894493103, "reward_total_composite_std": 0.007306722924113274} {"timestamp_utc": "2026-04-13T06:20:30Z", "mode": "train", "global_step": 3200, "epoch": 0.3214465092918132, "loss": 0.0043, "grad_norm": 8.837141990661621, "learning_rate": 3.0606060606060606e-07, "num_tokens": 6003236.0, "completions/mean_length": 30.75, "completions/min_length": 29.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.75, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.8888342976570129, "rewards/meter/std": 0.2593385577201843, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9615821838378906, "rewards/repeat_soft/std": 0.002596014179289341, "rewards/judge_quality/mean": 0.5349999666213989, "rewards/judge_quality/std": 0.24663449823856354, "rewards/total_composite/mean": 0.806633710861206, "rewards/total_composite/std": 0.09188058972358704, "reward": 0.806633710861206, "reward_std": 0.09188058972358704, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09330051392316818, "sampling/sampling_logp_difference/max": 1.433435082435608, "sampling/importance_sampling_ratio/min": 0.23848828673362732, "sampling/importance_sampling_ratio/mean": 1.016583800315857, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5086536146700382, "clip_ratio/low_mean": 0.04866518219932914, "clip_ratio/low_min": 0.04866518219932914, "clip_ratio/high_mean": 0.03996135760098696, "clip_ratio/high_max": 0.03996135760098696, "clip_ratio/region_mean": 0.0886265398003161, "reward_total_mean": 0.806633710861206, "reward_meter_mean": 0.8888342976570129, "reward_meter_std": 0.2593385577201843, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9615821838378906, "reward_repeat_soft_std": 0.002596014179289341, "reward_judge_quality_mean": 0.5349999666213989, "reward_judge_quality_std": 0.24663449823856354, "reward_total_composite_mean": 0.806633710861206, "reward_total_composite_std": 0.09188058972358704} {"timestamp_utc": "2026-04-13T06:21:19Z", "mode": "eval", "global_step": 3200, "epoch": 0.3214465092918132, "eval_loss": NaN, "eval_runtime": 49.2032, "eval_samples_per_second": 1.626, "eval_steps_per_second": 0.203, "eval_num_tokens": 6003236.0, "eval_completions/mean_length": 100.4, "eval_completions/min_length": 42.0, "eval_completions/max_length": 201.9, "eval_completions/clipped_ratio": 0.0125, "eval_completions/mean_terminated_length": 94.93928604125976, "eval_completions/min_terminated_length": 42.0, "eval_completions/max_terminated_length": 163.5, "eval_rewards/meter/mean": 0.9395426750183106, "eval_rewards/meter/std": 0.10366377299651504, "eval_rewards/count_adherence/mean": 0.9918750047683715, "eval_rewards/count_adherence/std": 0.02298096939921379, "eval_rewards/hard_gate/mean": 0.9875, "eval_rewards/hard_gate/std": 0.03535533845424652, "eval_rewards/repeat_soft/mean": 0.9003795146942138, "eval_rewards/repeat_soft/std": 0.08347392603754997, "eval_rewards/judge_quality/mean": 0.4444999933242798, "eval_rewards/judge_quality/std": 0.14541153237223625, "eval_rewards/total_composite/mean": 0.786187720298767, "eval_rewards/total_composite/std": 0.09351984802633524, "eval_reward": 0.786187720298767, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.037577960267663, "eval_sampling/sampling_logp_difference/max": 0.8591521978378296, "eval_sampling/importance_sampling_ratio/min": 0.43235677778720855, "eval_sampling/importance_sampling_ratio/mean": 1.0088049173355103, "eval_sampling/importance_sampling_ratio/max": 1.351186501979828, "eval_entropy": 0.3756983935832977, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.786187720298767, "eval_reward_meter_mean": 0.9395426750183106, "eval_reward_meter_std": 0.10366377299651504, "eval_reward_count_adherence_mean": 0.9918750047683715, "eval_reward_count_adherence_std": 0.02298096939921379, "eval_reward_hard_gate_mean": 0.9875, "eval_reward_hard_gate_std": 0.03535533845424652, "eval_reward_repeat_soft_mean": 0.9003795146942138, "eval_reward_repeat_soft_std": 0.08347392603754997, "eval_reward_judge_quality_mean": 0.4444999933242798, "eval_reward_judge_quality_std": 0.14541153237223625, "eval_reward_total_composite_mean": 0.786187720298767, "eval_reward_total_composite_std": 0.09351984802633524} {"timestamp_utc": "2026-04-13T06:21:29Z", "mode": "train", "global_step": 3201, "epoch": 0.32154696132596683, "loss": 0.0084, "grad_norm": 10.084071159362793, "learning_rate": 3.0303030303030305e-07, "num_tokens": 6005039.0, "completions/mean_length": 65.375, "completions/min_length": 63.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.375, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9852662682533264, "rewards/meter/std": 0.007031623739749193, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8867579698562622, "rewards/repeat_soft/std": 0.018294570967555046, "rewards/judge_quality/mean": 0.41749998927116394, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.807295560836792, "rewards/total_composite/std": 0.0197567418217659, "reward": 0.807295560836792, "reward_std": 0.019756726920604706, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07948412746191025, "sampling/sampling_logp_difference/max": 1.6859155893325806, "sampling/importance_sampling_ratio/min": 0.18527472019195557, "sampling/importance_sampling_ratio/mean": 1.0024940967559814, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.35487876646220684, "clip_ratio/low_mean": 0.021070076152682304, "clip_ratio/low_min": 0.021070076152682304, "clip_ratio/high_mean": 0.041717866668477654, "clip_ratio/high_max": 0.041717866668477654, "clip_ratio/region_mean": 0.06278794282115996, "reward_total_mean": 0.807295560836792, "reward_meter_mean": 0.9852662682533264, "reward_meter_std": 0.007031623739749193, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8867579698562622, "reward_repeat_soft_std": 0.018294570967555046, "reward_judge_quality_mean": 0.41749998927116394, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.807295560836792, "reward_total_composite_std": 0.0197567418217659} {"timestamp_utc": "2026-04-13T06:21:35Z", "mode": "train", "global_step": 3202, "epoch": 0.32164741336012054, "loss": 0.0216, "grad_norm": 8.265536308288574, "learning_rate": 3.0000000000000004e-07, "num_tokens": 6006755.0, "completions/mean_length": 64.5, "completions/min_length": 61.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.5, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9919717907905579, "rewards/meter/std": 0.0060705565847456455, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9638067483901978, "rewards/repeat_soft/std": 0.035584691911935806, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8187679648399353, "rewards/total_composite/std": 0.004968812223523855, "reward": 0.8187679648399353, "reward_std": 0.004968825727701187, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05966072902083397, "sampling/sampling_logp_difference/max": 1.2786505222320557, "sampling/importance_sampling_ratio/min": 0.27841275930404663, "sampling/importance_sampling_ratio/mean": 1.009598731994629, "sampling/importance_sampling_ratio/max": 1.9019572734832764, "entropy": 0.36375800147652626, "clip_ratio/low_mean": 0.009586247149854898, "clip_ratio/low_min": 0.009586247149854898, "clip_ratio/high_mean": 0.05438580468762666, "clip_ratio/high_max": 0.05438580468762666, "clip_ratio/region_mean": 0.06397205183748156, "reward_total_mean": 0.8187679648399353, "reward_meter_mean": 0.9919717907905579, "reward_meter_std": 0.0060705565847456455, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9638067483901978, "reward_repeat_soft_std": 0.035584691911935806, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8187679648399353, "reward_total_composite_std": 0.004968812223523855} {"timestamp_utc": "2026-04-13T06:21:47Z", "mode": "train", "global_step": 3203, "epoch": 0.32174786539427425, "loss": -0.1126, "grad_norm": 2.3935770988464355, "learning_rate": 2.96969696969697e-07, "num_tokens": 6008186.0, "completions/mean_length": 96.875, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 37.57143020629883, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.737946629524231, "rewards/meter/std": 0.32365474104881287, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9720765948295593, "rewards/repeat_soft/std": 0.03758621960878372, "rewards/judge_quality/mean": 0.36000001430511475, "rewards/judge_quality/std": 0.140813946723938, "rewards/total_composite/mean": 0.6428523063659668, "rewards/total_composite/std": 0.26878777146339417, "reward": 0.6428523063659668, "reward_std": 0.26878777146339417, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09563837200403214, "sampling/sampling_logp_difference/max": 1.5992586612701416, "sampling/importance_sampling_ratio/min": 0.2020462453365326, "sampling/importance_sampling_ratio/mean": 1.002284049987793, "sampling/importance_sampling_ratio/max": 1.7151890993118286, "entropy": 0.6219309568405151, "clip_ratio/low_mean": 0.010135134682059288, "clip_ratio/low_min": 0.010135134682059288, "clip_ratio/high_mean": 0.07568115531466901, "clip_ratio/high_max": 0.07568115531466901, "clip_ratio/region_mean": 0.0858162899967283, "reward_total_mean": 0.6428523063659668, "reward_meter_mean": 0.737946629524231, "reward_meter_std": 0.32365474104881287, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9720765948295593, "reward_repeat_soft_std": 0.03758621960878372, "reward_judge_quality_mean": 0.36000001430511475, "reward_judge_quality_std": 0.140813946723938, "reward_total_composite_mean": 0.6428523063659668, "reward_total_composite_std": 0.26878777146339417} {"timestamp_utc": "2026-04-13T06:21:58Z", "mode": "train", "global_step": 3204, "epoch": 0.3218483174284279, "loss": -0.1502, "grad_norm": 2.059568166732788, "learning_rate": 2.9393939393939397e-07, "num_tokens": 6009812.0, "completions/mean_length": 114.25, "completions/min_length": 54.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 57.42857360839844, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.8354671001434326, "rewards/meter/std": 0.34357112646102905, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9864230155944824, "rewards/repeat_soft/std": 0.011229261755943298, "rewards/judge_quality/mean": 0.45499998331069946, "rewards/judge_quality/std": 0.2334829568862915, "rewards/total_composite/mean": 0.7283324003219604, "rewards/total_composite/std": 0.30056318640708923, "reward": 0.7283324003219604, "reward_std": 0.30056318640708923, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09086567163467407, "sampling/sampling_logp_difference/max": 1.5202116966247559, "sampling/importance_sampling_ratio/min": 0.21866559982299805, "sampling/importance_sampling_ratio/mean": 1.0180771350860596, "sampling/importance_sampling_ratio/max": 1.980284333229065, "entropy": 0.39966892823576927, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06367313442751765, "clip_ratio/high_max": 0.06367313442751765, "clip_ratio/region_mean": 0.06367313442751765, "reward_total_mean": 0.7283324003219604, "reward_meter_mean": 0.8354671001434326, "reward_meter_std": 0.34357112646102905, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9864230155944824, "reward_repeat_soft_std": 0.011229261755943298, "reward_judge_quality_mean": 0.45499998331069946, "reward_judge_quality_std": 0.2334829568862915, "reward_total_composite_mean": 0.7283324003219604, "reward_total_composite_std": 0.30056318640708923} {"timestamp_utc": "2026-04-13T06:22:05Z", "mode": "train", "global_step": 3205, "epoch": 0.3219487694625816, "loss": 0.0244, "grad_norm": 10.488127708435059, "learning_rate": 2.9090909090909096e-07, "num_tokens": 6011518.0, "completions/mean_length": 56.25, "completions/min_length": 50.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.25, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9820189476013184, "rewards/meter/std": 0.01621944271028042, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9704910516738892, "rewards/repeat_soft/std": 0.029863344505429268, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.9087076187133789, "rewards/total_composite/std": 0.07405680418014526, "reward": 0.9087076187133789, "reward_std": 0.07405681163072586, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07965487241744995, "sampling/sampling_logp_difference/max": 1.477264404296875, "sampling/importance_sampling_ratio/min": 0.22826127707958221, "sampling/importance_sampling_ratio/mean": 1.0097990036010742, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44763942435383797, "clip_ratio/low_mean": 0.028906615916639566, "clip_ratio/low_min": 0.028906615916639566, "clip_ratio/high_mean": 0.04235917655751109, "clip_ratio/high_max": 0.04235917655751109, "clip_ratio/region_mean": 0.07126579247415066, "reward_total_mean": 0.9087076187133789, "reward_meter_mean": 0.9820189476013184, "reward_meter_std": 0.01621944271028042, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9704910516738892, "reward_repeat_soft_std": 0.029863344505429268, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.9087076187133789, "reward_total_composite_std": 0.07405680418014526} {"timestamp_utc": "2026-04-13T06:22:11Z", "mode": "train", "global_step": 3206, "epoch": 0.3220492214967353, "loss": 0.022, "grad_norm": 9.349202156066895, "learning_rate": 2.878787878787879e-07, "num_tokens": 6013268.0, "completions/mean_length": 55.75, "completions/min_length": 48.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.75, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.98261958360672, "rewards/meter/std": 0.015652742236852646, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9752628803253174, "rewards/repeat_soft/std": 0.015679694712162018, "rewards/judge_quality/mean": 0.5600000023841858, "rewards/judge_quality/std": 0.22258226573467255, "rewards/total_composite/mean": 0.8577051162719727, "rewards/total_composite/std": 0.06966240704059601, "reward": 0.8577051162719727, "reward_std": 0.0696624293923378, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08623603731393814, "sampling/sampling_logp_difference/max": 2.482630729675293, "sampling/importance_sampling_ratio/min": 0.0835232064127922, "sampling/importance_sampling_ratio/mean": 1.003678321838379, "sampling/importance_sampling_ratio/max": 1.8192311525344849, "entropy": 0.4140108786523342, "clip_ratio/low_mean": 0.06782087264582515, "clip_ratio/low_min": 0.06782087264582515, "clip_ratio/high_mean": 0.020067858509719372, "clip_ratio/high_max": 0.020067858509719372, "clip_ratio/region_mean": 0.08788873115554452, "reward_total_mean": 0.8577051162719727, "reward_meter_mean": 0.98261958360672, "reward_meter_std": 0.015652742236852646, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9752628803253174, "reward_repeat_soft_std": 0.015679694712162018, "reward_judge_quality_mean": 0.5600000023841858, "reward_judge_quality_std": 0.22258226573467255, "reward_total_composite_mean": 0.8577051162719727, "reward_total_composite_std": 0.06966240704059601} {"timestamp_utc": "2026-04-13T06:22:18Z", "mode": "train", "global_step": 3207, "epoch": 0.322149673530889, "loss": 0.0209, "grad_norm": 8.325886726379395, "learning_rate": 2.848484848484849e-07, "num_tokens": 6014991.0, "completions/mean_length": 59.375, "completions/min_length": 57.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.375, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9798913598060608, "rewards/meter/std": 0.012413685210049152, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9502273797988892, "rewards/repeat_soft/std": 0.037080783396959305, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.19078318774700165, "rewards/total_composite/mean": 0.8430988788604736, "rewards/total_composite/std": 0.051330018788576126, "reward": 0.8430988788604736, "reward_std": 0.05133001506328583, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06744074821472168, "sampling/sampling_logp_difference/max": 1.4706916809082031, "sampling/importance_sampling_ratio/min": 0.22976650297641754, "sampling/importance_sampling_ratio/mean": 1.009590744972229, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3039390780031681, "clip_ratio/low_mean": 0.0445174272172153, "clip_ratio/low_min": 0.0445174272172153, "clip_ratio/high_mean": 0.012934879399836063, "clip_ratio/high_max": 0.012934879399836063, "clip_ratio/region_mean": 0.05745230661705136, "reward_total_mean": 0.8430988788604736, "reward_meter_mean": 0.9798913598060608, "reward_meter_std": 0.012413685210049152, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9502273797988892, "reward_repeat_soft_std": 0.037080783396959305, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.19078318774700165, "reward_total_composite_mean": 0.8430988788604736, "reward_total_composite_std": 0.051330018788576126} {"timestamp_utc": "2026-04-13T06:22:30Z", "mode": "train", "global_step": 3208, "epoch": 0.3222501255650427, "loss": -0.1634, "grad_norm": 1.7474695444107056, "learning_rate": 2.818181818181819e-07, "num_tokens": 6016728.0, "completions/mean_length": 122.125, "completions/min_length": 63.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 66.42857360839844, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9544599056243896, "rewards/meter/std": 0.07304402440786362, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9647811055183411, "rewards/repeat_soft/std": 0.03871619701385498, "rewards/judge_quality/mean": 0.45249998569488525, "rewards/judge_quality/std": 0.21022097766399384, "rewards/total_composite/mean": 0.7353345155715942, "rewards/total_composite/std": 0.29994574189186096, "reward": 0.7353345155715942, "reward_std": 0.29994574189186096, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11629202216863632, "sampling/sampling_logp_difference/max": 5.891009330749512, "sampling/importance_sampling_ratio/min": 0.002764185192063451, "sampling/importance_sampling_ratio/mean": 1.0052462816238403, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48400160670280457, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09251904860138893, "clip_ratio/high_max": 0.09251904860138893, "clip_ratio/region_mean": 0.09251904860138893, "reward_total_mean": 0.7353345155715942, "reward_meter_mean": 0.9544599056243896, "reward_meter_std": 0.07304402440786362, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9647811055183411, "reward_repeat_soft_std": 0.03871619701385498, "reward_judge_quality_mean": 0.45249998569488525, "reward_judge_quality_std": 0.21022097766399384, "reward_total_composite_mean": 0.7353345155715942, "reward_total_composite_std": 0.29994574189186096} {"timestamp_utc": "2026-04-13T06:22:38Z", "mode": "train", "global_step": 3209, "epoch": 0.3223505775991964, "loss": 0.019, "grad_norm": 9.078673362731934, "learning_rate": 2.787878787878788e-07, "num_tokens": 6018431.0, "completions/mean_length": 67.875, "completions/min_length": 61.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.875, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.8987226486206055, "rewards/meter/std": 0.2329379916191101, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9720855951309204, "rewards/repeat_soft/std": 0.028417129069566727, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.8008837699890137, "rewards/total_composite/std": 0.12334685027599335, "reward": 0.8008837699890137, "reward_std": 0.12334683537483215, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09394552558660507, "sampling/sampling_logp_difference/max": 1.6495115756988525, "sampling/importance_sampling_ratio/min": 0.19214372336864471, "sampling/importance_sampling_ratio/mean": 1.010003924369812, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.46949149668216705, "clip_ratio/low_mean": 0.005514706019312143, "clip_ratio/low_min": 0.005514706019312143, "clip_ratio/high_mean": 0.0915129417553544, "clip_ratio/high_max": 0.0915129417553544, "clip_ratio/region_mean": 0.09702764777466655, "reward_total_mean": 0.8008837699890137, "reward_meter_mean": 0.8987226486206055, "reward_meter_std": 0.2329379916191101, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9720855951309204, "reward_repeat_soft_std": 0.028417129069566727, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.8008837699890137, "reward_total_composite_std": 0.12334685027599335} {"timestamp_utc": "2026-04-13T06:22:44Z", "mode": "train", "global_step": 3210, "epoch": 0.3224510296333501, "loss": 0.0113, "grad_norm": 13.226090431213379, "learning_rate": 2.757575757575758e-07, "num_tokens": 6020165.0, "completions/mean_length": 50.75, "completions/min_length": 45.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.75, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.8188774585723877, "rewards/meter/std": 0.1886228621006012, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9670810699462891, "rewards/repeat_soft/std": 0.026001283898949623, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7468279600143433, "rewards/total_composite/std": 0.08572742342948914, "reward": 0.7468279600143433, "reward_std": 0.08572742342948914, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10679169744253159, "sampling/sampling_logp_difference/max": 2.218684673309326, "sampling/importance_sampling_ratio/min": 0.1087520644068718, "sampling/importance_sampling_ratio/mean": 0.992428719997406, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44322941452264786, "clip_ratio/low_mean": 0.027115384582430124, "clip_ratio/low_min": 0.027115384582430124, "clip_ratio/high_mean": 0.06834371201694012, "clip_ratio/high_max": 0.06834371201694012, "clip_ratio/region_mean": 0.09545909659937024, "reward_total_mean": 0.7468279600143433, "reward_meter_mean": 0.8188774585723877, "reward_meter_std": 0.1886228621006012, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9670810699462891, "reward_repeat_soft_std": 0.026001283898949623, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7468279600143433, "reward_total_composite_std": 0.08572742342948914} {"timestamp_utc": "2026-04-13T06:22:50Z", "mode": "train", "global_step": 3211, "epoch": 0.32255148166750375, "loss": 0.0107, "grad_norm": 7.319581031799316, "learning_rate": 2.7272727272727274e-07, "num_tokens": 6021900.0, "completions/mean_length": 59.875, "completions/min_length": 56.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.875, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9619287252426147, "rewards/meter/std": 0.05526158958673477, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9516737461090088, "rewards/repeat_soft/std": 0.04132656753063202, "rewards/judge_quality/mean": 0.3725000023841858, "rewards/judge_quality/std": 0.11055056750774384, "rewards/total_composite/mean": 0.7897853255271912, "rewards/total_composite/std": 0.03420589491724968, "reward": 0.7897853255271912, "reward_std": 0.03420589491724968, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07319769263267517, "sampling/sampling_logp_difference/max": 1.275374412536621, "sampling/importance_sampling_ratio/min": 0.27932634949684143, "sampling/importance_sampling_ratio/mean": 1.0119316577911377, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39057549461722374, "clip_ratio/low_mean": 0.016666667303070426, "clip_ratio/low_min": 0.016666667303070426, "clip_ratio/high_mean": 0.047655189875513315, "clip_ratio/high_max": 0.047655189875513315, "clip_ratio/region_mean": 0.06432185717858374, "reward_total_mean": 0.7897853255271912, "reward_meter_mean": 0.9619287252426147, "reward_meter_std": 0.05526158958673477, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9516737461090088, "reward_repeat_soft_std": 0.04132656753063202, "reward_judge_quality_mean": 0.3725000023841858, "reward_judge_quality_std": 0.11055056750774384, "reward_total_composite_mean": 0.7897853255271912, "reward_total_composite_std": 0.03420589491724968} {"timestamp_utc": "2026-04-13T06:22:57Z", "mode": "train", "global_step": 3212, "epoch": 0.32265193370165746, "loss": 0.0317, "grad_norm": 14.715068817138672, "learning_rate": 2.6969696969696973e-07, "num_tokens": 6023516.0, "completions/mean_length": 33.0, "completions/min_length": 31.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9806780815124512, "rewards/meter/std": 0.02419968508183956, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9398219585418701, "rewards/repeat_soft/std": 0.034153103828430176, "rewards/judge_quality/mean": 0.6262500286102295, "rewards/judge_quality/std": 0.2432481348514557, "rewards/total_composite/mean": 0.8731623291969299, "rewards/total_composite/std": 0.07562211155891418, "reward": 0.8731623291969299, "reward_std": 0.07562211155891418, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11806738376617432, "sampling/sampling_logp_difference/max": 1.167313575744629, "sampling/importance_sampling_ratio/min": 0.3112018406391144, "sampling/importance_sampling_ratio/mean": 1.016494631767273, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5247361101210117, "clip_ratio/low_mean": 0.06415017577819526, "clip_ratio/low_min": 0.06415017577819526, "clip_ratio/high_mean": 0.03831005189567804, "clip_ratio/high_max": 0.03831005189567804, "clip_ratio/region_mean": 0.1024602276738733, "reward_total_mean": 0.8731623291969299, "reward_meter_mean": 0.9806780815124512, "reward_meter_std": 0.02419968508183956, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9398219585418701, "reward_repeat_soft_std": 0.034153103828430176, "reward_judge_quality_mean": 0.6262500286102295, "reward_judge_quality_std": 0.2432481348514557, "reward_total_composite_mean": 0.8731623291969299, "reward_total_composite_std": 0.07562211155891418} {"timestamp_utc": "2026-04-13T06:23:03Z", "mode": "train", "global_step": 3213, "epoch": 0.32275238573581116, "loss": 0.0851, "grad_norm": 15.21384048461914, "learning_rate": 2.666666666666667e-07, "num_tokens": 6025351.0, "completions/mean_length": 63.375, "completions/min_length": 57.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.375, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.7563471794128418, "rewards/meter/std": 0.41342997550964355, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9603686332702637, "rewards/repeat_soft/std": 0.047480564564466476, "rewards/judge_quality/mean": 0.5012500286102295, "rewards/judge_quality/std": 0.16974246501922607, "rewards/total_composite/mean": 0.7367681264877319, "rewards/total_composite/std": 0.20143736898899078, "reward": 0.7367681264877319, "reward_std": 0.20143736898899078, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13346216082572937, "sampling/sampling_logp_difference/max": 1.6394951343536377, "sampling/importance_sampling_ratio/min": 0.1940779983997345, "sampling/importance_sampling_ratio/mean": 1.0074552297592163, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5812714882194996, "clip_ratio/low_mean": 0.03041666653007269, "clip_ratio/low_min": 0.03041666653007269, "clip_ratio/high_mean": 0.09731007181107998, "clip_ratio/high_max": 0.09731007181107998, "clip_ratio/region_mean": 0.12772673834115267, "reward_total_mean": 0.7367681264877319, "reward_meter_mean": 0.7563471794128418, "reward_meter_std": 0.41342997550964355, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9603686332702637, "reward_repeat_soft_std": 0.047480564564466476, "reward_judge_quality_mean": 0.5012500286102295, "reward_judge_quality_std": 0.16974246501922607, "reward_total_composite_mean": 0.7367681264877319, "reward_total_composite_std": 0.20143736898899078} {"timestamp_utc": "2026-04-13T06:23:10Z", "mode": "train", "global_step": 3214, "epoch": 0.3228528377699648, "loss": 0.0366, "grad_norm": 43.91740798950195, "learning_rate": 2.6363636363636366e-07, "num_tokens": 6027011.0, "completions/mean_length": 44.5, "completions/min_length": 41.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.5, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.8015139102935791, "rewards/meter/std": 0.3368489146232605, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9780822992324829, "rewards/repeat_soft/std": 0.03582784906029701, "rewards/judge_quality/mean": 0.6200000047683716, "rewards/judge_quality/std": 0.22677870094776154, "rewards/total_composite/mean": 0.7944895029067993, "rewards/total_composite/std": 0.15963666141033173, "reward": 0.7944895029067993, "reward_std": 0.15963664650917053, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10518866032361984, "sampling/sampling_logp_difference/max": 3.7586233615875244, "sampling/importance_sampling_ratio/min": 0.02331581525504589, "sampling/importance_sampling_ratio/mean": 0.9945844411849976, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3463904745876789, "clip_ratio/low_mean": 0.03254241682589054, "clip_ratio/low_min": 0.03254241682589054, "clip_ratio/high_mean": 0.08324549999088049, "clip_ratio/high_max": 0.08324549999088049, "clip_ratio/region_mean": 0.11578791681677103, "reward_total_mean": 0.7944895029067993, "reward_meter_mean": 0.8015139102935791, "reward_meter_std": 0.3368489146232605, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9780822992324829, "reward_repeat_soft_std": 0.03582784906029701, "reward_judge_quality_mean": 0.6200000047683716, "reward_judge_quality_std": 0.22677870094776154, "reward_total_composite_mean": 0.7944895029067993, "reward_total_composite_std": 0.15963666141033173} {"timestamp_utc": "2026-04-13T06:23:17Z", "mode": "train", "global_step": 3215, "epoch": 0.3229532898041185, "loss": 0.0192, "grad_norm": 13.559785842895508, "learning_rate": 2.6060606060606065e-07, "num_tokens": 6028531.0, "completions/mean_length": 34.0, "completions/min_length": 31.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.641437292098999, "rewards/meter/std": 0.40649792551994324, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.6545218229293823, "rewards/total_composite/std": 0.18103906512260437, "reward": 0.6545218229293823, "reward_std": 0.18103906512260437, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10974156856536865, "sampling/sampling_logp_difference/max": 1.3207341432571411, "sampling/importance_sampling_ratio/min": 0.266939252614975, "sampling/importance_sampling_ratio/mean": 1.0040926933288574, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6603368930518627, "clip_ratio/low_mean": 0.036517622880637646, "clip_ratio/low_min": 0.036517622880637646, "clip_ratio/high_mean": 0.10316333174705505, "clip_ratio/high_max": 0.10316333174705505, "clip_ratio/region_mean": 0.1396809546276927, "reward_total_mean": 0.6545218229293823, "reward_meter_mean": 0.641437292098999, "reward_meter_std": 0.40649792551994324, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.6545218229293823, "reward_total_composite_std": 0.18103906512260437} {"timestamp_utc": "2026-04-13T06:23:24Z", "mode": "train", "global_step": 3216, "epoch": 0.32305374183827223, "loss": -0.0078, "grad_norm": 4.065310001373291, "learning_rate": 2.575757575757576e-07, "num_tokens": 6030919.0, "completions/mean_length": 136.5, "completions/min_length": 128.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.5, "completions/min_terminated_length": 128.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.9910476207733154, "rewards/meter/std": 0.0042385985143482685, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6793614625930786, "rewards/repeat_soft/std": 0.07342251390218735, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.8199076056480408, "rewards/total_composite/std": 0.055629849433898926, "reward": 0.8199076056480408, "reward_std": 0.05562985688447952, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06292369961738586, "sampling/sampling_logp_difference/max": 2.1026997566223145, "sampling/importance_sampling_ratio/min": 0.1221262738108635, "sampling/importance_sampling_ratio/mean": 1.0062676668167114, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31760179437696934, "clip_ratio/low_mean": 0.043631747364997864, "clip_ratio/low_min": 0.043631747364997864, "clip_ratio/high_mean": 0.01245915051549673, "clip_ratio/high_max": 0.01245915051549673, "clip_ratio/region_mean": 0.056090897880494595, "reward_total_mean": 0.8199076056480408, "reward_meter_mean": 0.9910476207733154, "reward_meter_std": 0.0042385985143482685, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6793614625930786, "reward_repeat_soft_std": 0.07342251390218735, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.8199076056480408, "reward_total_composite_std": 0.055629849433898926} {"timestamp_utc": "2026-04-13T06:23:31Z", "mode": "train", "global_step": 3217, "epoch": 0.32315419387242594, "loss": 0.032, "grad_norm": 6.7935051918029785, "learning_rate": 2.545454545454546e-07, "num_tokens": 6032735.0, "completions/mean_length": 68.0, "completions/min_length": 65.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9572833776473999, "rewards/meter/std": 0.0448995940387249, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8146108388900757, "rewards/repeat_soft/std": 0.08415432274341583, "rewards/judge_quality/mean": 0.2762500047683716, "rewards/judge_quality/std": 0.12603145837783813, "rewards/total_composite/mean": 0.7451136112213135, "rewards/total_composite/std": 0.03070583939552307, "reward": 0.7451136112213135, "reward_std": 0.030705822631716728, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07149244844913483, "sampling/sampling_logp_difference/max": 1.7347373962402344, "sampling/importance_sampling_ratio/min": 0.17644652724266052, "sampling/importance_sampling_ratio/mean": 1.0007601976394653, "sampling/importance_sampling_ratio/max": 1.9978201389312744, "entropy": 0.33451385609805584, "clip_ratio/low_mean": 0.036931168753653765, "clip_ratio/low_min": 0.036931168753653765, "clip_ratio/high_mean": 0.022732491604983807, "clip_ratio/high_max": 0.022732491604983807, "clip_ratio/region_mean": 0.05966366035863757, "reward_total_mean": 0.7451136112213135, "reward_meter_mean": 0.9572833776473999, "reward_meter_std": 0.0448995940387249, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8146108388900757, "reward_repeat_soft_std": 0.08415432274341583, "reward_judge_quality_mean": 0.2762500047683716, "reward_judge_quality_std": 0.12603145837783813, "reward_total_composite_mean": 0.7451136112213135, "reward_total_composite_std": 0.03070583939552307} {"timestamp_utc": "2026-04-13T06:23:38Z", "mode": "train", "global_step": 3218, "epoch": 0.3232546459065796, "loss": 0.0267, "grad_norm": 6.83361291885376, "learning_rate": 2.515151515151515e-07, "num_tokens": 6035106.0, "completions/mean_length": 128.375, "completions/min_length": 122.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.375, "completions/min_terminated_length": 122.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.7918075323104858, "rewards/meter/std": 0.26176917552948, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.922989010810852, "rewards/repeat_soft/std": 0.059653256088495255, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7152372598648071, "rewards/total_composite/std": 0.11008883267641068, "reward": 0.7152372598648071, "reward_std": 0.11008881777524948, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10096253454685211, "sampling/sampling_logp_difference/max": 2.0950675010681152, "sampling/importance_sampling_ratio/min": 0.12306193262338638, "sampling/importance_sampling_ratio/mean": 0.9999844431877136, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4825028032064438, "clip_ratio/low_mean": 0.027088341768831015, "clip_ratio/low_min": 0.027088341768831015, "clip_ratio/high_mean": 0.05824927240610123, "clip_ratio/high_max": 0.05824927240610123, "clip_ratio/region_mean": 0.08533761417493224, "reward_total_mean": 0.7152372598648071, "reward_meter_mean": 0.7918075323104858, "reward_meter_std": 0.26176917552948, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.922989010810852, "reward_repeat_soft_std": 0.059653256088495255, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7152372598648071, "reward_total_composite_std": 0.11008883267641068} {"timestamp_utc": "2026-04-13T06:23:49Z", "mode": "train", "global_step": 3219, "epoch": 0.3233550979407333, "loss": -0.1656, "grad_norm": 1.709477424621582, "learning_rate": 2.484848484848485e-07, "num_tokens": 6036980.0, "completions/mean_length": 123.25, "completions/min_length": 63.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 67.71428680419922, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.8481549024581909, "rewards/meter/std": 0.34685736894607544, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8854489326477051, "rewards/repeat_soft/std": 0.059989217668771744, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.6928396224975586, "rewards/total_composite/std": 0.2814231812953949, "reward": 0.6928396224975586, "reward_std": 0.2814231514930725, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08802459388971329, "sampling/sampling_logp_difference/max": 2.900266170501709, "sampling/importance_sampling_ratio/min": 0.055008575320243835, "sampling/importance_sampling_ratio/mean": 1.0054214000701904, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.30057143047451973, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0739822294563055, "clip_ratio/high_max": 0.0739822294563055, "clip_ratio/region_mean": 0.0739822294563055, "reward_total_mean": 0.6928396224975586, "reward_meter_mean": 0.8481549024581909, "reward_meter_std": 0.34685736894607544, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8854489326477051, "reward_repeat_soft_std": 0.059989217668771744, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.6928396224975586, "reward_total_composite_std": 0.2814231812953949} {"timestamp_utc": "2026-04-13T06:23:56Z", "mode": "train", "global_step": 3220, "epoch": 0.323455549974887, "loss": 0.0789, "grad_norm": 8.304230690002441, "learning_rate": 2.4545454545454545e-07, "num_tokens": 6038948.0, "completions/mean_length": 88.0, "completions/min_length": 81.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 88.0, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9885962009429932, "rewards/meter/std": 0.006467185448855162, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.932789146900177, "rewards/repeat_soft/std": 0.04801025986671448, "rewards/judge_quality/mean": 0.39750000834465027, "rewards/judge_quality/std": 0.15745748579502106, "rewards/total_composite/mean": 0.8073972463607788, "rewards/total_composite/std": 0.047888681292533875, "reward": 0.8073972463607788, "reward_std": 0.047888681292533875, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09059111773967743, "sampling/sampling_logp_difference/max": 1.4618186950683594, "sampling/importance_sampling_ratio/min": 0.23181429505348206, "sampling/importance_sampling_ratio/mean": 0.9996791481971741, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5004241243004799, "clip_ratio/low_mean": 0.03620381187647581, "clip_ratio/low_min": 0.03620381187647581, "clip_ratio/high_mean": 0.06549616763368249, "clip_ratio/high_max": 0.06549616763368249, "clip_ratio/region_mean": 0.1016999795101583, "reward_total_mean": 0.8073972463607788, "reward_meter_mean": 0.9885962009429932, "reward_meter_std": 0.006467185448855162, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.932789146900177, "reward_repeat_soft_std": 0.04801025986671448, "reward_judge_quality_mean": 0.39750000834465027, "reward_judge_quality_std": 0.15745748579502106, "reward_total_composite_mean": 0.8073972463607788, "reward_total_composite_std": 0.047888681292533875} {"timestamp_utc": "2026-04-13T06:24:08Z", "mode": "train", "global_step": 3221, "epoch": 0.32355600200904067, "loss": -0.2064, "grad_norm": 1.8610773086547852, "learning_rate": 2.4242424242424244e-07, "num_tokens": 6041076.0, "completions/mean_length": 162.0, "completions/min_length": 103.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 112.00000762939453, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.952502965927124, "rewards/meter/std": 0.1073024794459343, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.935093879699707, "rewards/repeat_soft/std": 0.0508669838309288, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.1403057724237442, "rewards/total_composite/mean": 0.6838349103927612, "rewards/total_composite/std": 0.2809838652610779, "reward": 0.6838349103927612, "reward_std": 0.2809838652610779, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0833519697189331, "sampling/sampling_logp_difference/max": 1.887386441230774, "sampling/importance_sampling_ratio/min": 0.15146715939044952, "sampling/importance_sampling_ratio/mean": 1.0138475894927979, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.373858492821455, "clip_ratio/low_mean": 0.006756756920367479, "clip_ratio/low_min": 0.006756756920367479, "clip_ratio/high_mean": 0.06551532074809074, "clip_ratio/high_max": 0.06551532074809074, "clip_ratio/region_mean": 0.07227207766845822, "reward_total_mean": 0.6838349103927612, "reward_meter_mean": 0.952502965927124, "reward_meter_std": 0.1073024794459343, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.2651650309562683, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.935093879699707, "reward_repeat_soft_std": 0.0508669838309288, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.1403057724237442, "reward_total_composite_mean": 0.6838349103927612, "reward_total_composite_std": 0.2809838652610779} {"timestamp_utc": "2026-04-13T06:24:15Z", "mode": "train", "global_step": 3222, "epoch": 0.3236564540431944, "loss": -0.0011, "grad_norm": 5.3915605545043945, "learning_rate": 2.3939393939393943e-07, "num_tokens": 6043337.0, "completions/mean_length": 122.625, "completions/min_length": 112.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.625, "completions/min_terminated_length": 112.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.9751021265983582, "rewards/meter/std": 0.04980902001261711, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9188723564147949, "rewards/repeat_soft/std": 0.05797549709677696, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.0843462198972702, "rewards/total_composite/mean": 0.7914957404136658, "rewards/total_composite/std": 0.03895266726613045, "reward": 0.7914957404136658, "reward_std": 0.03895265981554985, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0738486722111702, "sampling/sampling_logp_difference/max": 1.3404560089111328, "sampling/importance_sampling_ratio/min": 0.2617262899875641, "sampling/importance_sampling_ratio/mean": 1.0087991952896118, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4419437274336815, "clip_ratio/low_mean": 0.02210886008106172, "clip_ratio/low_min": 0.02210886008106172, "clip_ratio/high_mean": 0.05243017617613077, "clip_ratio/high_max": 0.05243017617613077, "clip_ratio/region_mean": 0.07453903625719249, "reward_total_mean": 0.7914957404136658, "reward_meter_mean": 0.9751021265983582, "reward_meter_std": 0.04980902001261711, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9188723564147949, "reward_repeat_soft_std": 0.05797549709677696, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.0843462198972702, "reward_total_composite_mean": 0.7914957404136658, "reward_total_composite_std": 0.03895266726613045} {"timestamp_utc": "2026-04-13T06:24:22Z", "mode": "train", "global_step": 3223, "epoch": 0.3237569060773481, "loss": 0.0466, "grad_norm": 10.445274353027344, "learning_rate": 2.3636363636363637e-07, "num_tokens": 6045186.0, "completions/mean_length": 67.125, "completions/min_length": 64.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.125, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9850270748138428, "rewards/meter/std": 0.006309048738330603, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.895771861076355, "rewards/repeat_soft/std": 0.08944407850503922, "rewards/judge_quality/mean": 0.3962499797344208, "rewards/judge_quality/std": 0.09085899591445923, "rewards/total_composite/mean": 0.8017143607139587, "rewards/total_composite/std": 0.026749050244688988, "reward": 0.8017143607139587, "reward_std": 0.02674904279410839, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08895958960056305, "sampling/sampling_logp_difference/max": 1.167954921722412, "sampling/importance_sampling_ratio/min": 0.31100231409072876, "sampling/importance_sampling_ratio/mean": 0.9892621636390686, "sampling/importance_sampling_ratio/max": 1.77684485912323, "entropy": 0.4830439127981663, "clip_ratio/low_mean": 0.012663399102166295, "clip_ratio/low_min": 0.012663399102166295, "clip_ratio/high_mean": 0.07090717693790793, "clip_ratio/high_max": 0.07090717693790793, "clip_ratio/region_mean": 0.08357057604007423, "reward_total_mean": 0.8017143607139587, "reward_meter_mean": 0.9850270748138428, "reward_meter_std": 0.006309048738330603, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.895771861076355, "reward_repeat_soft_std": 0.08944407850503922, "reward_judge_quality_mean": 0.3962499797344208, "reward_judge_quality_std": 0.09085899591445923, "reward_total_composite_mean": 0.8017143607139587, "reward_total_composite_std": 0.026749050244688988} {"timestamp_utc": "2026-04-13T06:24:28Z", "mode": "train", "global_step": 3224, "epoch": 0.32385735811150174, "loss": 0.0416, "grad_norm": 8.142398834228516, "learning_rate": 2.3333333333333336e-07, "num_tokens": 6047006.0, "completions/mean_length": 54.5, "completions/min_length": 52.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.5, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9772263765335083, "rewards/meter/std": 0.019112776964902878, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9702147245407104, "rewards/repeat_soft/std": 0.032313235104084015, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8161482810974121, "rewards/total_composite/std": 0.01267396192997694, "reward": 0.8161482810974121, "reward_std": 0.012673964723944664, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08946435898542404, "sampling/sampling_logp_difference/max": 1.172399640083313, "sampling/importance_sampling_ratio/min": 0.3096230626106262, "sampling/importance_sampling_ratio/mean": 0.9972015619277954, "sampling/importance_sampling_ratio/max": 1.7104859352111816, "entropy": 0.4341824948787689, "clip_ratio/low_mean": 0.0042372881434857845, "clip_ratio/low_min": 0.0042372881434857845, "clip_ratio/high_mean": 0.0858296777587384, "clip_ratio/high_max": 0.0858296777587384, "clip_ratio/region_mean": 0.09006696590222418, "reward_total_mean": 0.8161482810974121, "reward_meter_mean": 0.9772263765335083, "reward_meter_std": 0.019112776964902878, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9702147245407104, "reward_repeat_soft_std": 0.032313235104084015, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8161482810974121, "reward_total_composite_std": 0.01267396192997694} {"timestamp_utc": "2026-04-13T06:24:35Z", "mode": "train", "global_step": 3225, "epoch": 0.32395781014565544, "loss": 0.0146, "grad_norm": 10.446993827819824, "learning_rate": 2.3030303030303032e-07, "num_tokens": 6048600.0, "completions/mean_length": 44.25, "completions/min_length": 41.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.25, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.9141412377357483, "rewards/meter/std": 0.08431105315685272, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8124046325683594, "rewards/repeat_soft/std": 0.15710239112377167, "rewards/judge_quality/mean": 0.39249998331069946, "rewards/judge_quality/std": 0.08892211318016052, "rewards/total_composite/mean": 0.7603539824485779, "rewards/total_composite/std": 0.042521748691797256, "reward": 0.7603539824485779, "reward_std": 0.042521748691797256, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07969329506158829, "sampling/sampling_logp_difference/max": 1.1045646667480469, "sampling/importance_sampling_ratio/min": 0.33135509490966797, "sampling/importance_sampling_ratio/mean": 1.004123568534851, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43592431396245956, "clip_ratio/low_mean": 0.023695904528722167, "clip_ratio/low_min": 0.023695904528722167, "clip_ratio/high_mean": 0.03835658123716712, "clip_ratio/high_max": 0.03835658123716712, "clip_ratio/region_mean": 0.06205248576588929, "reward_total_mean": 0.7603539824485779, "reward_meter_mean": 0.9141412377357483, "reward_meter_std": 0.08431105315685272, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8124046325683594, "reward_repeat_soft_std": 0.15710239112377167, "reward_judge_quality_mean": 0.39249998331069946, "reward_judge_quality_std": 0.08892211318016052, "reward_total_composite_mean": 0.7603539824485779, "reward_total_composite_std": 0.042521748691797256} {"timestamp_utc": "2026-04-13T06:24:41Z", "mode": "train", "global_step": 3226, "epoch": 0.32405826217980915, "loss": 0.0369, "grad_norm": 14.179804801940918, "learning_rate": 2.2727272727272729e-07, "num_tokens": 6050376.0, "completions/mean_length": 59.0, "completions/min_length": 54.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.8709956407546997, "rewards/meter/std": 0.2535388767719269, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9030297994613647, "rewards/repeat_soft/std": 0.046673376113176346, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7616260051727295, "rewards/total_composite/std": 0.1117570549249649, "reward": 0.7616260051727295, "reward_std": 0.11175703257322311, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0742662325501442, "sampling/sampling_logp_difference/max": 2.3773529529571533, "sampling/importance_sampling_ratio/min": 0.09279588609933853, "sampling/importance_sampling_ratio/mean": 1.0080952644348145, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3506549522280693, "clip_ratio/low_mean": 0.008196720853447914, "clip_ratio/low_min": 0.008196720853447914, "clip_ratio/high_mean": 0.04093195591121912, "clip_ratio/high_max": 0.04093195591121912, "clip_ratio/region_mean": 0.049128676764667034, "reward_total_mean": 0.7616260051727295, "reward_meter_mean": 0.8709956407546997, "reward_meter_std": 0.2535388767719269, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9030297994613647, "reward_repeat_soft_std": 0.046673376113176346, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7616260051727295, "reward_total_composite_std": 0.1117570549249649} {"timestamp_utc": "2026-04-13T06:24:47Z", "mode": "train", "global_step": 3227, "epoch": 0.3241587142139628, "loss": 0.0183, "grad_norm": 7.388851165771484, "learning_rate": 2.2424242424242425e-07, "num_tokens": 6052056.0, "completions/mean_length": 45.0, "completions/min_length": 44.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.0, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9706989526748657, "rewards/meter/std": 0.004953043535351753, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9568074345588684, "rewards/repeat_soft/std": 0.03928307443857193, "rewards/judge_quality/mean": 0.25, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7574952840805054, "rewards/total_composite/std": 0.0030405709985643625, "reward": 0.7574952840805054, "reward_std": 0.0030405856668949127, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0718625858426094, "sampling/sampling_logp_difference/max": 1.3643134832382202, "sampling/importance_sampling_ratio/min": 0.2555560767650604, "sampling/importance_sampling_ratio/mean": 0.9817526936531067, "sampling/importance_sampling_ratio/max": 1.7485792636871338, "entropy": 0.29772803746163845, "clip_ratio/low_mean": 0.008459596196189523, "clip_ratio/low_min": 0.008459596196189523, "clip_ratio/high_mean": 0.036182478070259094, "clip_ratio/high_max": 0.036182478070259094, "clip_ratio/region_mean": 0.04464207426644862, "reward_total_mean": 0.7574952840805054, "reward_meter_mean": 0.9706989526748657, "reward_meter_std": 0.004953043535351753, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9568074345588684, "reward_repeat_soft_std": 0.03928307443857193, "reward_judge_quality_mean": 0.25, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7574952840805054, "reward_total_composite_std": 0.0030405709985643625} {"timestamp_utc": "2026-04-13T06:24:54Z", "mode": "train", "global_step": 3228, "epoch": 0.3242591662481165, "loss": 0.0076, "grad_norm": 5.196498870849609, "learning_rate": 2.2121212121212124e-07, "num_tokens": 6053635.0, "completions/mean_length": 44.375, "completions/min_length": 43.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.375, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9731504321098328, "rewards/meter/std": 0.002959948033094406, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9243773221969604, "rewards/repeat_soft/std": 0.07665771245956421, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8074804544448853, "rewards/total_composite/std": 0.007533563766628504, "reward": 0.8074804544448853, "reward_std": 0.007533560041338205, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.045568350702524185, "sampling/sampling_logp_difference/max": 0.9723063707351685, "sampling/importance_sampling_ratio/min": 0.3782097399234772, "sampling/importance_sampling_ratio/mean": 1.0050443410873413, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.22370882146060467, "clip_ratio/low_mean": 0.017045455053448677, "clip_ratio/low_min": 0.017045455053448677, "clip_ratio/high_mean": 0.027830999344587326, "clip_ratio/high_max": 0.027830999344587326, "clip_ratio/region_mean": 0.044876454398036, "reward_total_mean": 0.8074804544448853, "reward_meter_mean": 0.9731504321098328, "reward_meter_std": 0.002959948033094406, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9243773221969604, "reward_repeat_soft_std": 0.07665771245956421, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8074804544448853, "reward_total_composite_std": 0.007533563766628504} {"timestamp_utc": "2026-04-13T06:25:00Z", "mode": "train", "global_step": 3229, "epoch": 0.3243596182822702, "loss": 0.0187, "grad_norm": 9.86949634552002, "learning_rate": 2.181818181818182e-07, "num_tokens": 6055255.0, "completions/mean_length": 42.5, "completions/min_length": 40.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.5, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.9718329906463623, "rewards/meter/std": 0.008836928755044937, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9257901310920715, "rewards/repeat_soft/std": 0.1082187071442604, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8115288615226746, "rewards/total_composite/std": 0.015496993437409401, "reward": 0.8115288615226746, "reward_std": 0.015496988780796528, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07562746107578278, "sampling/sampling_logp_difference/max": 2.0585124492645264, "sampling/importance_sampling_ratio/min": 0.12764370441436768, "sampling/importance_sampling_ratio/mean": 0.9902325868606567, "sampling/importance_sampling_ratio/max": 1.7333264350891113, "entropy": 0.283459035679698, "clip_ratio/low_mean": 0.0317156205419451, "clip_ratio/low_min": 0.0317156205419451, "clip_ratio/high_mean": 0.056024113204330206, "clip_ratio/high_max": 0.056024113204330206, "clip_ratio/region_mean": 0.0877397337462753, "reward_total_mean": 0.8115288615226746, "reward_meter_mean": 0.9718329906463623, "reward_meter_std": 0.008836928755044937, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9257901310920715, "reward_repeat_soft_std": 0.1082187071442604, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8115288615226746, "reward_total_composite_std": 0.015496993437409401} {"timestamp_utc": "2026-04-13T06:25:07Z", "mode": "train", "global_step": 3230, "epoch": 0.32446007031642393, "loss": -0.0106, "grad_norm": 4.679012298583984, "learning_rate": 2.1515151515151517e-07, "num_tokens": 6057564.0, "completions/mean_length": 118.625, "completions/min_length": 108.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.625, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.9747105836868286, "rewards/meter/std": 0.029704507440328598, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9169681072235107, "rewards/repeat_soft/std": 0.030787404626607895, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.8288165330886841, "rewards/total_composite/std": 0.047527458518743515, "reward": 0.8288165330886841, "reward_std": 0.04752745479345322, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07517785578966141, "sampling/sampling_logp_difference/max": 4.940582275390625, "sampling/importance_sampling_ratio/min": 0.007150433491915464, "sampling/importance_sampling_ratio/mean": 0.9966751933097839, "sampling/importance_sampling_ratio/max": 1.9977366924285889, "entropy": 0.376058891415596, "clip_ratio/low_mean": 0.04414787422865629, "clip_ratio/low_min": 0.04414787422865629, "clip_ratio/high_mean": 0.027033126913011074, "clip_ratio/high_max": 0.027033126913011074, "clip_ratio/region_mean": 0.07118100114166737, "reward_total_mean": 0.8288165330886841, "reward_meter_mean": 0.9747105836868286, "reward_meter_std": 0.029704507440328598, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9169681072235107, "reward_repeat_soft_std": 0.030787404626607895, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.8288165330886841, "reward_total_composite_std": 0.047527458518743515} {"timestamp_utc": "2026-04-13T06:25:18Z", "mode": "train", "global_step": 3231, "epoch": 0.3245605223505776, "loss": -0.1448, "grad_norm": 1.6610002517700195, "learning_rate": 2.1212121212121216e-07, "num_tokens": 6059102.0, "completions/mean_length": 112.25, "completions/min_length": 50.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 55.142860412597656, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.8677184581756592, "rewards/meter/std": 0.34000924229621887, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9708826541900635, "rewards/repeat_soft/std": 0.027740824967622757, "rewards/judge_quality/mean": 0.5174999833106995, "rewards/judge_quality/std": 0.2841905951499939, "rewards/total_composite/mean": 0.7582037448883057, "rewards/total_composite/std": 0.31261834502220154, "reward": 0.7582037448883057, "reward_std": 0.31261831521987915, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07546006143093109, "sampling/sampling_logp_difference/max": 1.080902099609375, "sampling/importance_sampling_ratio/min": 0.3392893075942993, "sampling/importance_sampling_ratio/mean": 1.0288935899734497, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3936598300933838, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06504504848271608, "clip_ratio/high_max": 0.06504504848271608, "clip_ratio/region_mean": 0.06504504848271608, "reward_total_mean": 0.7582037448883057, "reward_meter_mean": 0.8677184581756592, "reward_meter_std": 0.34000924229621887, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9708826541900635, "reward_repeat_soft_std": 0.027740824967622757, "reward_judge_quality_mean": 0.5174999833106995, "reward_judge_quality_std": 0.2841905951499939, "reward_total_composite_mean": 0.7582037448883057, "reward_total_composite_std": 0.31261834502220154} {"timestamp_utc": "2026-04-13T06:25:26Z", "mode": "train", "global_step": 3232, "epoch": 0.3246609743847313, "loss": 0.0494, "grad_norm": 7.378695964813232, "learning_rate": 2.090909090909091e-07, "num_tokens": 6061581.0, "completions/mean_length": 126.875, "completions/min_length": 115.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.875, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.9188686609268188, "rewards/meter/std": 0.09790186583995819, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9116958975791931, "rewards/repeat_soft/std": 0.0502464734017849, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.7615355253219604, "rewards/total_composite/std": 0.052723731845617294, "reward": 0.7615355253219604, "reward_std": 0.052723728120326996, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09810690581798553, "sampling/sampling_logp_difference/max": 2.6587371826171875, "sampling/importance_sampling_ratio/min": 0.07003661245107651, "sampling/importance_sampling_ratio/mean": 1.0096195936203003, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4616267792880535, "clip_ratio/low_mean": 0.045985568314790726, "clip_ratio/low_min": 0.045985568314790726, "clip_ratio/high_mean": 0.05317143350839615, "clip_ratio/high_max": 0.05317143350839615, "clip_ratio/region_mean": 0.09915700182318687, "reward_total_mean": 0.7615355253219604, "reward_meter_mean": 0.9188686609268188, "reward_meter_std": 0.09790186583995819, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9116958975791931, "reward_repeat_soft_std": 0.0502464734017849, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.7615355253219604, "reward_total_composite_std": 0.052723731845617294} {"timestamp_utc": "2026-04-13T06:25:33Z", "mode": "train", "global_step": 3233, "epoch": 0.324761426418885, "loss": -0.0123, "grad_norm": 6.095747470855713, "learning_rate": 2.060606060606061e-07, "num_tokens": 6064104.0, "completions/mean_length": 135.375, "completions/min_length": 115.0, "completions/max_length": 154.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 135.375, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 154.0, "rewards/meter/mean": 0.8432003855705261, "rewards/meter/std": 0.18774321675300598, "rewards/count_adherence/mean": 0.8500000238418579, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7172197103500366, "rewards/repeat_soft/std": 0.091814324259758, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.20860078930854797, "rewards/total_composite/mean": 0.7106621265411377, "rewards/total_composite/std": 0.10465390980243683, "reward": 0.7106621265411377, "reward_std": 0.10465391725301743, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0419699102640152, "sampling/sampling_logp_difference/max": 1.5883629322052002, "sampling/importance_sampling_ratio/min": 0.2042597234249115, "sampling/importance_sampling_ratio/mean": 1.0014065504074097, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.20795171335339546, "clip_ratio/low_mean": 0.012743965489789844, "clip_ratio/low_min": 0.012743965489789844, "clip_ratio/high_mean": 0.025731497444212437, "clip_ratio/high_max": 0.025731497444212437, "clip_ratio/region_mean": 0.03847546293400228, "reward_total_mean": 0.7106621265411377, "reward_meter_mean": 0.8432003855705261, "reward_meter_std": 0.18774321675300598, "reward_count_adherence_mean": 0.8500000238418579, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7172197103500366, "reward_repeat_soft_std": 0.091814324259758, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.20860078930854797, "reward_total_composite_mean": 0.7106621265411377, "reward_total_composite_std": 0.10465390980243683} {"timestamp_utc": "2026-04-13T06:25:41Z", "mode": "train", "global_step": 3234, "epoch": 0.32486187845303865, "loss": 0.0044, "grad_norm": 12.587029457092285, "learning_rate": 2.0303030303030303e-07, "num_tokens": 6067055.0, "completions/mean_length": 160.875, "completions/min_length": 152.0, "completions/max_length": 170.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 160.875, "completions/min_terminated_length": 152.0, "completions/max_terminated_length": 170.0, "rewards/meter/mean": 0.9547489285469055, "rewards/meter/std": 0.10377105325460434, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9503442049026489, "rewards/repeat_soft/std": 0.02516033500432968, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.781546413898468, "rewards/total_composite/std": 0.045340053737163544, "reward": 0.781546413898468, "reward_std": 0.04534006118774414, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10714346915483475, "sampling/sampling_logp_difference/max": 1.985379934310913, "sampling/importance_sampling_ratio/min": 0.13732841610908508, "sampling/importance_sampling_ratio/mean": 0.9963619112968445, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3800734914839268, "clip_ratio/low_mean": 0.050106339156627655, "clip_ratio/low_min": 0.050106339156627655, "clip_ratio/high_mean": 0.04490234889090061, "clip_ratio/high_max": 0.04490234889090061, "clip_ratio/region_mean": 0.09500868804752827, "reward_total_mean": 0.781546413898468, "reward_meter_mean": 0.9547489285469055, "reward_meter_std": 0.10377105325460434, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9503442049026489, "reward_repeat_soft_std": 0.02516033500432968, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.781546413898468, "reward_total_composite_std": 0.045340053737163544} {"timestamp_utc": "2026-04-13T06:25:47Z", "mode": "train", "global_step": 3235, "epoch": 0.32496233048719236, "loss": 0.0084, "grad_norm": 9.554847717285156, "learning_rate": 2.0000000000000002e-07, "num_tokens": 6068742.0, "completions/mean_length": 52.875, "completions/min_length": 49.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.875, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9691773653030396, "rewards/meter/std": 0.03309769555926323, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9457270503044128, "rewards/repeat_soft/std": 0.03625684231519699, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8078275322914124, "rewards/total_composite/std": 0.014160392805933952, "reward": 0.8078275322914124, "reward_std": 0.014160390011966228, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0675663948059082, "sampling/sampling_logp_difference/max": 1.5113525390625, "sampling/importance_sampling_ratio/min": 0.22061139345169067, "sampling/importance_sampling_ratio/mean": 1.0177537202835083, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3443086985498667, "clip_ratio/low_mean": 0.017148525919765234, "clip_ratio/low_min": 0.017148525919765234, "clip_ratio/high_mean": 0.03795844782143831, "clip_ratio/high_max": 0.03795844782143831, "clip_ratio/region_mean": 0.055106973741203547, "reward_total_mean": 0.8078275322914124, "reward_meter_mean": 0.9691773653030396, "reward_meter_std": 0.03309769555926323, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9457270503044128, "reward_repeat_soft_std": 0.03625684231519699, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8078275322914124, "reward_total_composite_std": 0.014160392805933952} {"timestamp_utc": "2026-04-13T06:25:54Z", "mode": "train", "global_step": 3236, "epoch": 0.32506278252134607, "loss": 0.0374, "grad_norm": 8.903108596801758, "learning_rate": 1.9696969696969698e-07, "num_tokens": 6070600.0, "completions/mean_length": 61.25, "completions/min_length": 54.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.25, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9698642492294312, "rewards/meter/std": 0.028874430805444717, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9587895274162292, "rewards/repeat_soft/std": 0.03717319294810295, "rewards/judge_quality/mean": 0.6137499809265137, "rewards/judge_quality/std": 0.24318939447402954, "rewards/total_composite/mean": 0.866442859172821, "rewards/total_composite/std": 0.07624262571334839, "reward": 0.866442859172821, "reward_std": 0.0762426108121872, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10631377249956131, "sampling/sampling_logp_difference/max": 2.018042802810669, "sampling/importance_sampling_ratio/min": 0.13291536271572113, "sampling/importance_sampling_ratio/mean": 1.0034184455871582, "sampling/importance_sampling_ratio/max": 1.9886243343353271, "entropy": 0.5713762007653713, "clip_ratio/low_mean": 0.0742348488420248, "clip_ratio/low_min": 0.0742348488420248, "clip_ratio/high_mean": 0.05710565485060215, "clip_ratio/high_max": 0.05710565485060215, "clip_ratio/region_mean": 0.13134050369262695, "reward_total_mean": 0.866442859172821, "reward_meter_mean": 0.9698642492294312, "reward_meter_std": 0.028874430805444717, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9587895274162292, "reward_repeat_soft_std": 0.03717319294810295, "reward_judge_quality_mean": 0.6137499809265137, "reward_judge_quality_std": 0.24318939447402954, "reward_total_composite_mean": 0.866442859172821, "reward_total_composite_std": 0.07624262571334839} {"timestamp_utc": "2026-04-13T06:26:00Z", "mode": "train", "global_step": 3237, "epoch": 0.3251632345554997, "loss": 0.0408, "grad_norm": 13.685219764709473, "learning_rate": 1.9393939393939395e-07, "num_tokens": 6072310.0, "completions/mean_length": 44.75, "completions/min_length": 40.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.75, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.6087789535522461, "rewards/meter/std": 0.341485857963562, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9727563858032227, "rewards/repeat_soft/std": 0.023228151723742485, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.6419761180877686, "rewards/total_composite/std": 0.16840477287769318, "reward": 0.6419761180877686, "reward_std": 0.16840477287769318, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10196728259325027, "sampling/sampling_logp_difference/max": 1.5499916076660156, "sampling/importance_sampling_ratio/min": 0.212249755859375, "sampling/importance_sampling_ratio/mean": 0.9870988726615906, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4232148081064224, "clip_ratio/low_mean": 0.019752275431528687, "clip_ratio/low_min": 0.019752275431528687, "clip_ratio/high_mean": 0.048536534421145916, "clip_ratio/high_max": 0.048536534421145916, "clip_ratio/region_mean": 0.0682888098526746, "reward_total_mean": 0.6419761180877686, "reward_meter_mean": 0.6087789535522461, "reward_meter_std": 0.341485857963562, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9727563858032227, "reward_repeat_soft_std": 0.023228151723742485, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.6419761180877686, "reward_total_composite_std": 0.16840477287769318} {"timestamp_utc": "2026-04-13T06:26:11Z", "mode": "train", "global_step": 3238, "epoch": 0.32526368658965343, "loss": -0.1566, "grad_norm": 1.4064173698425293, "learning_rate": 1.9090909090909094e-07, "num_tokens": 6073948.0, "completions/mean_length": 119.75, "completions/min_length": 58.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 63.71428680419922, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9266701340675354, "rewards/meter/std": 0.19176128506660461, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9492105841636658, "rewards/repeat_soft/std": 0.046853963285684586, "rewards/judge_quality/mean": 0.3499999940395355, "rewards/judge_quality/std": 0.15445756912231445, "rewards/total_composite/mean": 0.7107095122337341, "rewards/total_composite/std": 0.2877958118915558, "reward": 0.7107095122337341, "reward_std": 0.2877957820892334, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11300832778215408, "sampling/sampling_logp_difference/max": 1.546102523803711, "sampling/importance_sampling_ratio/min": 0.21307681500911713, "sampling/importance_sampling_ratio/mean": 0.9894124865531921, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4389031417667866, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09733521193265915, "clip_ratio/high_max": 0.09733521193265915, "clip_ratio/region_mean": 0.09733521193265915, "reward_total_mean": 0.7107095122337341, "reward_meter_mean": 0.9266701340675354, "reward_meter_std": 0.19176128506660461, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9492105841636658, "reward_repeat_soft_std": 0.046853963285684586, "reward_judge_quality_mean": 0.3499999940395355, "reward_judge_quality_std": 0.15445756912231445, "reward_total_composite_mean": 0.7107095122337341, "reward_total_composite_std": 0.2877958118915558} {"timestamp_utc": "2026-04-13T06:26:18Z", "mode": "train", "global_step": 3239, "epoch": 0.32536413862380714, "loss": -0.0078, "grad_norm": 10.98618221282959, "learning_rate": 1.878787878787879e-07, "num_tokens": 6075593.0, "completions/mean_length": 42.625, "completions/min_length": 40.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.625, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.8699852228164673, "rewards/meter/std": 0.2604580521583557, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8620626926422119, "rewards/repeat_soft/std": 0.09860948473215103, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.7581996321678162, "rewards/total_composite/std": 0.11484934389591217, "reward": 0.7581996321678162, "reward_std": 0.11484935134649277, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.051758863031864166, "sampling/sampling_logp_difference/max": 1.91534423828125, "sampling/importance_sampling_ratio/min": 0.1472911238670349, "sampling/importance_sampling_ratio/mean": 1.003432273864746, "sampling/importance_sampling_ratio/max": 1.6732558012008667, "entropy": 0.25247687101364136, "clip_ratio/low_mean": 0.0031250000465661287, "clip_ratio/low_min": 0.0031250000465661287, "clip_ratio/high_mean": 0.0474384194239974, "clip_ratio/high_max": 0.0474384194239974, "clip_ratio/region_mean": 0.05056341947056353, "reward_total_mean": 0.7581996321678162, "reward_meter_mean": 0.8699852228164673, "reward_meter_std": 0.2604580521583557, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8620626926422119, "reward_repeat_soft_std": 0.09860948473215103, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.7581996321678162, "reward_total_composite_std": 0.11484934389591217} {"timestamp_utc": "2026-04-13T06:26:25Z", "mode": "train", "global_step": 3240, "epoch": 0.32546459065796085, "loss": 0.0017, "grad_norm": 5.8944220542907715, "learning_rate": 1.8484848484848486e-07, "num_tokens": 6078306.0, "completions/mean_length": 144.125, "completions/min_length": 137.0, "completions/max_length": 155.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 144.125, "completions/min_terminated_length": 137.0, "completions/max_terminated_length": 155.0, "rewards/meter/mean": 0.9888542890548706, "rewards/meter/std": 0.009772595018148422, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9112828969955444, "rewards/repeat_soft/std": 0.03953630104660988, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8121126890182495, "rewards/total_composite/std": 0.003991363570094109, "reward": 0.8121126890182495, "reward_std": 0.003991360310465097, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08419448882341385, "sampling/sampling_logp_difference/max": 1.5500377416610718, "sampling/importance_sampling_ratio/min": 0.21223996579647064, "sampling/importance_sampling_ratio/mean": 0.9945482611656189, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3381131887435913, "clip_ratio/low_mean": 0.027408655267208815, "clip_ratio/low_min": 0.027408655267208815, "clip_ratio/high_mean": 0.05123586766421795, "clip_ratio/high_max": 0.05123586766421795, "clip_ratio/region_mean": 0.07864452293142676, "reward_total_mean": 0.8121126890182495, "reward_meter_mean": 0.9888542890548706, "reward_meter_std": 0.009772595018148422, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9112828969955444, "reward_repeat_soft_std": 0.03953630104660988, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8121126890182495, "reward_total_composite_std": 0.003991363570094109} {"timestamp_utc": "2026-04-13T06:26:31Z", "mode": "train", "global_step": 3241, "epoch": 0.3255650426921145, "loss": 0.0714, "grad_norm": 15.671236038208008, "learning_rate": 1.8181818181818183e-07, "num_tokens": 6079843.0, "completions/mean_length": 42.125, "completions/min_length": 36.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.125, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.7425194382667542, "rewards/meter/std": 0.3110401928424835, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9684605598449707, "rewards/repeat_soft/std": 0.04590519890189171, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.7279797792434692, "rewards/total_composite/std": 0.15345704555511475, "reward": 0.7279797792434692, "reward_std": 0.15345704555511475, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11163033545017242, "sampling/sampling_logp_difference/max": 1.8572888374328613, "sampling/importance_sampling_ratio/min": 0.15609526634216309, "sampling/importance_sampling_ratio/mean": 1.0031501054763794, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.34712741896510124, "clip_ratio/low_mean": 0.039107789285480976, "clip_ratio/low_min": 0.039107789285480976, "clip_ratio/high_mean": 0.06508335494436324, "clip_ratio/high_max": 0.06508335494436324, "clip_ratio/region_mean": 0.10419114422984421, "reward_total_mean": 0.7279797792434692, "reward_meter_mean": 0.7425194382667542, "reward_meter_std": 0.3110401928424835, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9684605598449707, "reward_repeat_soft_std": 0.04590519890189171, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.7279797792434692, "reward_total_composite_std": 0.15345704555511475} {"timestamp_utc": "2026-04-13T06:26:37Z", "mode": "train", "global_step": 3242, "epoch": 0.3256654947262682, "loss": 0.0312, "grad_norm": 9.444353103637695, "learning_rate": 1.7878787878787882e-07, "num_tokens": 6081802.0, "completions/mean_length": 65.875, "completions/min_length": 60.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.875, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9913788437843323, "rewards/meter/std": 0.005941360257565975, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9772897958755493, "rewards/repeat_soft/std": 0.02114529348909855, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.894849419593811, "rewards/total_composite/std": 0.0804687887430191, "reward": 0.894849419593811, "reward_std": 0.0804687887430191, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08007072657346725, "sampling/sampling_logp_difference/max": 1.5817327499389648, "sampling/importance_sampling_ratio/min": 0.2056185007095337, "sampling/importance_sampling_ratio/mean": 0.9963533878326416, "sampling/importance_sampling_ratio/max": 1.8406896591186523, "entropy": 0.35855041071772575, "clip_ratio/low_mean": 0.03819699818268418, "clip_ratio/low_min": 0.03819699818268418, "clip_ratio/high_mean": 0.04753229534253478, "clip_ratio/high_max": 0.04753229534253478, "clip_ratio/region_mean": 0.08572929352521896, "reward_total_mean": 0.894849419593811, "reward_meter_mean": 0.9913788437843323, "reward_meter_std": 0.005941360257565975, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9772897958755493, "reward_repeat_soft_std": 0.02114529348909855, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.894849419593811, "reward_total_composite_std": 0.0804687887430191} {"timestamp_utc": "2026-04-13T06:26:45Z", "mode": "train", "global_step": 3243, "epoch": 0.3257659467604219, "loss": 0.0176, "grad_norm": 5.027962684631348, "learning_rate": 1.7575757575757576e-07, "num_tokens": 6084596.0, "completions/mean_length": 158.25, "completions/min_length": 146.0, "completions/max_length": 166.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 158.25, "completions/min_terminated_length": 146.0, "completions/max_terminated_length": 166.0, "rewards/meter/mean": 0.9411507844924927, "rewards/meter/std": 0.12081427872180939, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9029760360717773, "rewards/repeat_soft/std": 0.06316275894641876, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.25150617957115173, "rewards/total_composite/mean": 0.8209404349327087, "rewards/total_composite/std": 0.111705482006073, "reward": 0.8209404349327087, "reward_std": 0.11170550435781479, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08082151412963867, "sampling/sampling_logp_difference/max": 2.8333663940429688, "sampling/importance_sampling_ratio/min": 0.058814529329538345, "sampling/importance_sampling_ratio/mean": 1.0038326978683472, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3819093108177185, "clip_ratio/low_mean": 0.05294918501749635, "clip_ratio/low_min": 0.05294918501749635, "clip_ratio/high_mean": 0.026560315862298012, "clip_ratio/high_max": 0.026560315862298012, "clip_ratio/region_mean": 0.07950950087979436, "reward_total_mean": 0.8209404349327087, "reward_meter_mean": 0.9411507844924927, "reward_meter_std": 0.12081427872180939, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9029760360717773, "reward_repeat_soft_std": 0.06316275894641876, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.25150617957115173, "reward_total_composite_mean": 0.8209404349327087, "reward_total_composite_std": 0.111705482006073} {"timestamp_utc": "2026-04-13T06:26:52Z", "mode": "train", "global_step": 3244, "epoch": 0.3258663987945756, "loss": -0.026, "grad_norm": 4.846555709838867, "learning_rate": 1.7272727272727275e-07, "num_tokens": 6087462.0, "completions/mean_length": 150.25, "completions/min_length": 133.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 150.25, "completions/min_terminated_length": 133.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.9926590919494629, "rewards/meter/std": 0.0038543047849088907, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9307045936584473, "rewards/repeat_soft/std": 0.030043909326195717, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8120170831680298, "rewards/total_composite/std": 0.009101489558815956, "reward": 0.8120170831680298, "reward_std": 0.009101488627493382, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08545925468206406, "sampling/sampling_logp_difference/max": 1.9702987670898438, "sampling/importance_sampling_ratio/min": 0.1394152045249939, "sampling/importance_sampling_ratio/mean": 0.9975674748420715, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3601324036717415, "clip_ratio/low_mean": 0.017812388017773628, "clip_ratio/low_min": 0.017812388017773628, "clip_ratio/high_mean": 0.06956185167655349, "clip_ratio/high_max": 0.06956185167655349, "clip_ratio/region_mean": 0.08737423969432712, "reward_total_mean": 0.8120170831680298, "reward_meter_mean": 0.9926590919494629, "reward_meter_std": 0.0038543047849088907, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9307045936584473, "reward_repeat_soft_std": 0.030043909326195717, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8120170831680298, "reward_total_composite_std": 0.009101489558815956} {"timestamp_utc": "2026-04-13T06:27:04Z", "mode": "train", "global_step": 3245, "epoch": 0.3259668508287293, "loss": -0.1941, "grad_norm": 1.5017858743667603, "learning_rate": 1.6969696969696974e-07, "num_tokens": 6089476.0, "completions/mean_length": 144.75, "completions/min_length": 85.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 92.28572082519531, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.870286226272583, "rewards/meter/std": 0.33268263936042786, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9698172211647034, "rewards/repeat_soft/std": 0.01664149761199951, "rewards/judge_quality/mean": 0.3462499976158142, "rewards/judge_quality/std": 0.15202796459197998, "rewards/total_composite/mean": 0.7090479135513306, "rewards/total_composite/std": 0.28705278038978577, "reward": 0.7090479135513306, "reward_std": 0.28705278038978577, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0851241871714592, "sampling/sampling_logp_difference/max": 1.2963509559631348, "sampling/importance_sampling_ratio/min": 0.2735280990600586, "sampling/importance_sampling_ratio/mean": 1.0117872953414917, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3885704390704632, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08822135627269745, "clip_ratio/high_max": 0.08822135627269745, "clip_ratio/region_mean": 0.08822135627269745, "reward_total_mean": 0.7090479135513306, "reward_meter_mean": 0.870286226272583, "reward_meter_std": 0.33268263936042786, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9698172211647034, "reward_repeat_soft_std": 0.01664149761199951, "reward_judge_quality_mean": 0.3462499976158142, "reward_judge_quality_std": 0.15202796459197998, "reward_total_composite_mean": 0.7090479135513306, "reward_total_composite_std": 0.28705278038978577} {"timestamp_utc": "2026-04-13T06:27:11Z", "mode": "train", "global_step": 3246, "epoch": 0.326067302862883, "loss": 0.0309, "grad_norm": 5.520633220672607, "learning_rate": 1.6666666666666668e-07, "num_tokens": 6091691.0, "completions/mean_length": 106.875, "completions/min_length": 102.0, "completions/max_length": 113.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.875, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.9867696166038513, "rewards/meter/std": 0.004479403141885996, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6680517196655273, "rewards/repeat_soft/std": 0.1137128472328186, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.7576014995574951, "rewards/total_composite/std": 0.03977473825216293, "reward": 0.7576014995574951, "reward_std": 0.03977472335100174, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06681662052869797, "sampling/sampling_logp_difference/max": 2.0340821743011475, "sampling/importance_sampling_ratio/min": 0.13080048561096191, "sampling/importance_sampling_ratio/mean": 1.008666753768921, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3497532606124878, "clip_ratio/low_mean": 0.02434374950826168, "clip_ratio/low_min": 0.02434374950826168, "clip_ratio/high_mean": 0.031963180750608444, "clip_ratio/high_max": 0.031963180750608444, "clip_ratio/region_mean": 0.056306930258870125, "reward_total_mean": 0.7576014995574951, "reward_meter_mean": 0.9867696166038513, "reward_meter_std": 0.004479403141885996, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6680517196655273, "reward_repeat_soft_std": 0.1137128472328186, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.7576014995574951, "reward_total_composite_std": 0.03977473825216293} {"timestamp_utc": "2026-04-13T06:27:18Z", "mode": "train", "global_step": 3247, "epoch": 0.32616775489703664, "loss": -0.0284, "grad_norm": 5.175244331359863, "learning_rate": 1.6363636363636367e-07, "num_tokens": 6094355.0, "completions/mean_length": 152.0, "completions/min_length": 139.0, "completions/max_length": 173.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 152.0, "completions/min_terminated_length": 139.0, "completions/max_terminated_length": 173.0, "rewards/meter/mean": 0.8020341396331787, "rewards/meter/std": 0.33246317505836487, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7889448404312134, "rewards/repeat_soft/std": 0.09171022474765778, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.7281848192214966, "rewards/total_composite/std": 0.1605508029460907, "reward": 0.7281848192214966, "reward_std": 0.1605508178472519, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05993976071476936, "sampling/sampling_logp_difference/max": 1.5194826126098633, "sampling/importance_sampling_ratio/min": 0.2188250869512558, "sampling/importance_sampling_ratio/mean": 1.0047589540481567, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3042035289108753, "clip_ratio/low_mean": 0.018789058551192284, "clip_ratio/low_min": 0.018789058551192284, "clip_ratio/high_mean": 0.04141288925893605, "clip_ratio/high_max": 0.04141288925893605, "clip_ratio/region_mean": 0.06020194781012833, "reward_total_mean": 0.7281848192214966, "reward_meter_mean": 0.8020341396331787, "reward_meter_std": 0.33246317505836487, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7889448404312134, "reward_repeat_soft_std": 0.09171022474765778, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.7281848192214966, "reward_total_composite_std": 0.1605508029460907} {"timestamp_utc": "2026-04-13T06:27:25Z", "mode": "train", "global_step": 3248, "epoch": 0.32626820693119035, "loss": 0.0143, "grad_norm": 10.865485191345215, "learning_rate": 1.606060606060606e-07, "num_tokens": 6095944.0, "completions/mean_length": 48.625, "completions/min_length": 40.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.625, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9908843040466309, "rewards/meter/std": 0.004043751861900091, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8540040254592896, "rewards/repeat_soft/std": 0.05973939225077629, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.8042982816696167, "rewards/total_composite/std": 0.025501776486635208, "reward": 0.8042982816696167, "reward_std": 0.025501780211925507, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07956909388303757, "sampling/sampling_logp_difference/max": 1.1593972444534302, "sampling/importance_sampling_ratio/min": 0.3136752247810364, "sampling/importance_sampling_ratio/mean": 1.0217499732971191, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37428712844848633, "clip_ratio/low_mean": 0.010204081423580647, "clip_ratio/low_min": 0.010204081423580647, "clip_ratio/high_mean": 0.07119420799426734, "clip_ratio/high_max": 0.07119420799426734, "clip_ratio/region_mean": 0.08139828941784799, "reward_total_mean": 0.8042982816696167, "reward_meter_mean": 0.9908843040466309, "reward_meter_std": 0.004043751861900091, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8540040254592896, "reward_repeat_soft_std": 0.05973939225077629, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.8042982816696167, "reward_total_composite_std": 0.025501776486635208} {"timestamp_utc": "2026-04-13T06:27:31Z", "mode": "train", "global_step": 3249, "epoch": 0.32636865896534406, "loss": 0.0451, "grad_norm": 9.368904113769531, "learning_rate": 1.575757575757576e-07, "num_tokens": 6097791.0, "completions/mean_length": 61.875, "completions/min_length": 53.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.875, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9887232780456543, "rewards/meter/std": 0.01159851998090744, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8669333457946777, "rewards/repeat_soft/std": 0.1014527902007103, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.8034937977790833, "rewards/total_composite/std": 0.021207725629210472, "reward": 0.8034937977790833, "reward_std": 0.021207723766565323, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09251075983047485, "sampling/sampling_logp_difference/max": 1.863739013671875, "sampling/importance_sampling_ratio/min": 0.15509165823459625, "sampling/importance_sampling_ratio/mean": 0.9953866004943848, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40526995807886124, "clip_ratio/low_mean": 0.009527972200885415, "clip_ratio/low_min": 0.009527972200885415, "clip_ratio/high_mean": 0.05847551999613643, "clip_ratio/high_max": 0.05847551999613643, "clip_ratio/region_mean": 0.06800349219702184, "reward_total_mean": 0.8034937977790833, "reward_meter_mean": 0.9887232780456543, "reward_meter_std": 0.01159851998090744, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8669333457946777, "reward_repeat_soft_std": 0.1014527902007103, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.8034937977790833, "reward_total_composite_std": 0.021207725629210472} {"timestamp_utc": "2026-04-13T06:27:38Z", "mode": "train", "global_step": 3250, "epoch": 0.3264691109994977, "loss": 0.0395, "grad_norm": 6.4319844245910645, "learning_rate": 1.5454545454545456e-07, "num_tokens": 6099747.0, "completions/mean_length": 71.5, "completions/min_length": 69.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.5, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9866569638252258, "rewards/meter/std": 0.0052825999446213245, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9778780341148376, "rewards/repeat_soft/std": 0.023104384541511536, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.8365334272384644, "rewards/total_composite/std": 0.054186802357435226, "reward": 0.8365334272384644, "reward_std": 0.054186806082725525, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10866546630859375, "sampling/sampling_logp_difference/max": 1.585984706878662, "sampling/importance_sampling_ratio/min": 0.2047460675239563, "sampling/importance_sampling_ratio/mean": 1.0015995502471924, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5896763540804386, "clip_ratio/low_mean": 0.08898999448865652, "clip_ratio/low_min": 0.08898999448865652, "clip_ratio/high_mean": 0.023550724610686302, "clip_ratio/high_max": 0.023550724610686302, "clip_ratio/region_mean": 0.11254071909934282, "reward_total_mean": 0.8365334272384644, "reward_meter_mean": 0.9866569638252258, "reward_meter_std": 0.0052825999446213245, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9778780341148376, "reward_repeat_soft_std": 0.023104384541511536, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.8365334272384644, "reward_total_composite_std": 0.054186802357435226} {"timestamp_utc": "2026-04-13T06:28:36Z", "mode": "eval", "global_step": 3250, "epoch": 0.3264691109994977, "eval_loss": NaN, "eval_runtime": 57.8796, "eval_samples_per_second": 1.382, "eval_steps_per_second": 0.173, "eval_num_tokens": 6099747.0, "eval_completions/mean_length": 112.375, "eval_completions/min_length": 41.0, "eval_completions/max_length": 270.4, "eval_completions/clipped_ratio": 0.0375, "eval_completions/mean_terminated_length": 96.56428680419921, "eval_completions/min_terminated_length": 41.0, "eval_completions/max_terminated_length": 165.3, "eval_rewards/meter/mean": 0.9297292649745941, "eval_rewards/meter/std": 0.11818294869735838, "eval_rewards/count_adherence/mean": 0.9908333241939544, "eval_rewards/count_adherence/std": 0.021043315529823303, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.07071067690849304, "eval_rewards/repeat_soft/mean": 0.9073995471000671, "eval_rewards/repeat_soft/std": 0.07370126023888587, "eval_rewards/judge_quality/mean": 0.4101249992847443, "eval_rewards/judge_quality/std": 0.14334231093525887, "eval_rewards/total_composite/mean": 0.7633032500743866, "eval_rewards/total_composite/std": 0.12393471878021955, "eval_reward": 0.7633032500743866, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.03875854723155499, "eval_sampling/sampling_logp_difference/max": 0.9185259819030762, "eval_sampling/importance_sampling_ratio/min": 0.4036492109298706, "eval_sampling/importance_sampling_ratio/mean": 1.006665575504303, "eval_sampling/importance_sampling_ratio/max": 1.3250461578369142, "eval_entropy": 0.3729119151830673, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7633032500743866, "eval_reward_meter_mean": 0.9297292649745941, "eval_reward_meter_std": 0.11818294869735838, "eval_reward_count_adherence_mean": 0.9908333241939544, "eval_reward_count_adherence_std": 0.021043315529823303, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.07071067690849304, "eval_reward_repeat_soft_mean": 0.9073995471000671, "eval_reward_repeat_soft_std": 0.07370126023888587, "eval_reward_judge_quality_mean": 0.4101249992847443, "eval_reward_judge_quality_std": 0.14334231093525887, "eval_reward_total_composite_mean": 0.7633032500743866, "eval_reward_total_composite_std": 0.12393471878021955} {"timestamp_utc": "2026-04-13T06:28:45Z", "mode": "train", "global_step": 3251, "epoch": 0.3265695630336514, "loss": 0.0216, "grad_norm": 6.853144645690918, "learning_rate": 1.5151515151515152e-07, "num_tokens": 6101393.0, "completions/mean_length": 44.75, "completions/min_length": 42.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.75, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.7547199726104736, "rewards/meter/std": 0.23552055656909943, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9357906579971313, "rewards/repeat_soft/std": 0.08558089286088943, "rewards/judge_quality/mean": 0.39375001192092896, "rewards/judge_quality/std": 0.09941794723272324, "rewards/total_composite/mean": 0.7013280391693115, "rewards/total_composite/std": 0.10205158591270447, "reward": 0.7013280391693115, "reward_std": 0.10205157846212387, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07187296450138092, "sampling/sampling_logp_difference/max": 1.1486923694610596, "sampling/importance_sampling_ratio/min": 0.31705108284950256, "sampling/importance_sampling_ratio/mean": 1.0073719024658203, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3614392839372158, "clip_ratio/low_mean": 0.014204545877873898, "clip_ratio/low_min": 0.014204545877873898, "clip_ratio/high_mean": 0.031107661314308643, "clip_ratio/high_max": 0.031107661314308643, "clip_ratio/region_mean": 0.04531220719218254, "reward_total_mean": 0.7013280391693115, "reward_meter_mean": 0.7547199726104736, "reward_meter_std": 0.23552055656909943, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9357906579971313, "reward_repeat_soft_std": 0.08558089286088943, "reward_judge_quality_mean": 0.39375001192092896, "reward_judge_quality_std": 0.09941794723272324, "reward_total_composite_mean": 0.7013280391693115, "reward_total_composite_std": 0.10205158591270447} {"timestamp_utc": "2026-04-13T06:28:52Z", "mode": "train", "global_step": 3252, "epoch": 0.32667001506780513, "loss": 0.0539, "grad_norm": 8.908750534057617, "learning_rate": 1.484848484848485e-07, "num_tokens": 6103055.0, "completions/mean_length": 56.75, "completions/min_length": 54.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.75, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.990951657295227, "rewards/meter/std": 0.002783421892672777, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9502529501914978, "rewards/repeat_soft/std": 0.03482326492667198, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404937028885, "rewards/total_composite/mean": 0.8578285574913025, "rewards/total_composite/std": 0.06763999164104462, "reward": 0.8578285574913025, "reward_std": 0.06763999164104462, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09582863748073578, "sampling/sampling_logp_difference/max": 1.7175579071044922, "sampling/importance_sampling_ratio/min": 0.17950399219989777, "sampling/importance_sampling_ratio/mean": 1.0081651210784912, "sampling/importance_sampling_ratio/max": 1.9691970348358154, "entropy": 0.4345022886991501, "clip_ratio/low_mean": 0.058671738021075726, "clip_ratio/low_min": 0.058671738021075726, "clip_ratio/high_mean": 0.03181818127632141, "clip_ratio/high_max": 0.03181818127632141, "clip_ratio/region_mean": 0.09048991929739714, "reward_total_mean": 0.8578285574913025, "reward_meter_mean": 0.990951657295227, "reward_meter_std": 0.002783421892672777, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9502529501914978, "reward_repeat_soft_std": 0.03482326492667198, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404937028885, "reward_total_composite_mean": 0.8578285574913025, "reward_total_composite_std": 0.06763999164104462} {"timestamp_utc": "2026-04-13T06:29:04Z", "mode": "train", "global_step": 3253, "epoch": 0.32677046710195884, "loss": -0.0546, "grad_norm": 2.3542799949645996, "learning_rate": 1.4545454545454548e-07, "num_tokens": 6104540.0, "completions/mean_length": 151.625, "completions/min_length": 26.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 31.5, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.6157086491584778, "rewards/meter/std": 0.5099570155143738, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.4629100561141968, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9605984687805176, "rewards/repeat_soft/std": 0.03929799422621727, "rewards/judge_quality/mean": 0.2462500035762787, "rewards/judge_quality/std": 0.17270433902740479, "rewards/total_composite/mean": 0.4842034578323364, "rewards/total_composite/std": 0.4021008610725403, "reward": 0.4842034578323364, "reward_std": 0.4021008312702179, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09290111809968948, "sampling/sampling_logp_difference/max": 0.734032154083252, "sampling/importance_sampling_ratio/min": 0.4799697995185852, "sampling/importance_sampling_ratio/mean": 1.0322684049606323, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5326561219990253, "clip_ratio/low_mean": 0.014285714365541935, "clip_ratio/low_min": 0.014285714365541935, "clip_ratio/high_mean": 0.036221210611984134, "clip_ratio/high_max": 0.036221210611984134, "clip_ratio/region_mean": 0.05050692497752607, "reward_total_mean": 0.4842034578323364, "reward_meter_mean": 0.6157086491584778, "reward_meter_std": 0.5099570155143738, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.4629100561141968, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9605984687805176, "reward_repeat_soft_std": 0.03929799422621727, "reward_judge_quality_mean": 0.2462500035762787, "reward_judge_quality_std": 0.17270433902740479, "reward_total_composite_mean": 0.4842034578323364, "reward_total_composite_std": 0.4021008610725403} {"timestamp_utc": "2026-04-13T06:29:11Z", "mode": "train", "global_step": 3254, "epoch": 0.3268709191361125, "loss": 0.0301, "grad_norm": 4.474758625030518, "learning_rate": 1.4242424242424244e-07, "num_tokens": 6106944.0, "completions/mean_length": 130.5, "completions/min_length": 122.0, "completions/max_length": 141.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 130.5, "completions/min_terminated_length": 122.0, "completions/max_terminated_length": 141.0, "rewards/meter/mean": 0.990755021572113, "rewards/meter/std": 0.00326615315862, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8876057863235474, "rewards/repeat_soft/std": 0.09746994078159332, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.7877253293991089, "rewards/total_composite/std": 0.03961410000920296, "reward": 0.7877253293991089, "reward_std": 0.03961408883333206, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07181194424629211, "sampling/sampling_logp_difference/max": 1.7602672576904297, "sampling/importance_sampling_ratio/min": 0.17199888825416565, "sampling/importance_sampling_ratio/mean": 1.007485032081604, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3988274112343788, "clip_ratio/low_mean": 0.023798524867743254, "clip_ratio/low_min": 0.023798524867743254, "clip_ratio/high_mean": 0.04348963359370828, "clip_ratio/high_max": 0.04348963359370828, "clip_ratio/region_mean": 0.06728815846145153, "reward_total_mean": 0.7877253293991089, "reward_meter_mean": 0.990755021572113, "reward_meter_std": 0.00326615315862, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8876057863235474, "reward_repeat_soft_std": 0.09746994078159332, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.7877253293991089, "reward_total_composite_std": 0.03961410000920296} {"timestamp_utc": "2026-04-13T06:29:18Z", "mode": "train", "global_step": 3255, "epoch": 0.3269713711702662, "loss": -0.0055, "grad_norm": 7.509485244750977, "learning_rate": 1.393939393939394e-07, "num_tokens": 6109038.0, "completions/mean_length": 92.75, "completions/min_length": 80.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.75, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.9187684059143066, "rewards/meter/std": 0.07268919050693512, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9189883470535278, "rewards/repeat_soft/std": 0.048996590077877045, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.7749696373939514, "rewards/total_composite/std": 0.02967192232608795, "reward": 0.7749696373939514, "reward_std": 0.029671920463442802, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09491462260484695, "sampling/sampling_logp_difference/max": 1.1375504732131958, "sampling/importance_sampling_ratio/min": 0.3206034004688263, "sampling/importance_sampling_ratio/mean": 1.0124822854995728, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6131033152341843, "clip_ratio/low_mean": 0.04306775191798806, "clip_ratio/low_min": 0.04306775191798806, "clip_ratio/high_mean": 0.05735392216593027, "clip_ratio/high_max": 0.05735392216593027, "clip_ratio/region_mean": 0.10042167408391833, "reward_total_mean": 0.7749696373939514, "reward_meter_mean": 0.9187684059143066, "reward_meter_std": 0.07268919050693512, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9189883470535278, "reward_repeat_soft_std": 0.048996590077877045, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.7749696373939514, "reward_total_composite_std": 0.02967192232608795} {"timestamp_utc": "2026-04-13T06:29:24Z", "mode": "train", "global_step": 3256, "epoch": 0.3270718232044199, "loss": 0.0629, "grad_norm": 14.71054458618164, "learning_rate": 1.3636363636363637e-07, "num_tokens": 6110468.0, "completions/mean_length": 27.75, "completions/min_length": 26.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.75, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9848361015319824, "rewards/meter/std": 0.010830932296812534, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8210512399673462, "rewards/total_composite/std": 0.0059755714610219, "reward": 0.8210512399673462, "reward_std": 0.005975569598376751, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08612169325351715, "sampling/sampling_logp_difference/max": 0.9375901222229004, "sampling/importance_sampling_ratio/min": 0.3915703296661377, "sampling/importance_sampling_ratio/mean": 1.0150221586227417, "sampling/importance_sampling_ratio/max": 1.7175008058547974, "entropy": 0.3783934451639652, "clip_ratio/low_mean": 0.07242063526064157, "clip_ratio/low_min": 0.07242063526064157, "clip_ratio/high_mean": 0.04221357451751828, "clip_ratio/high_max": 0.04221357451751828, "clip_ratio/region_mean": 0.11463420977815986, "reward_total_mean": 0.8210512399673462, "reward_meter_mean": 0.9848361015319824, "reward_meter_std": 0.010830932296812534, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8210512399673462, "reward_total_composite_std": 0.0059755714610219} {"timestamp_utc": "2026-04-13T06:29:31Z", "mode": "train", "global_step": 3257, "epoch": 0.32717227523857356, "loss": 0.0203, "grad_norm": 6.475704193115234, "learning_rate": 1.3333333333333336e-07, "num_tokens": 6112510.0, "completions/mean_length": 89.25, "completions/min_length": 78.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.25, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.9283384084701538, "rewards/meter/std": 0.09005939960479736, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8968449831008911, "rewards/repeat_soft/std": 0.03951030597090721, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.81343674659729, "rewards/total_composite/std": 0.08030222356319427, "reward": 0.81343674659729, "reward_std": 0.08030222356319427, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07081729918718338, "sampling/sampling_logp_difference/max": 3.757415533065796, "sampling/importance_sampling_ratio/min": 0.023343995213508606, "sampling/importance_sampling_ratio/mean": 1.0112913846969604, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36570291966199875, "clip_ratio/low_mean": 0.057600845117121935, "clip_ratio/low_min": 0.057600845117121935, "clip_ratio/high_mean": 0.026116155553609133, "clip_ratio/high_max": 0.026116155553609133, "clip_ratio/region_mean": 0.08371700067073107, "reward_total_mean": 0.81343674659729, "reward_meter_mean": 0.9283384084701538, "reward_meter_std": 0.09005939960479736, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8968449831008911, "reward_repeat_soft_std": 0.03951030597090721, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.81343674659729, "reward_total_composite_std": 0.08030222356319427} {"timestamp_utc": "2026-04-13T06:29:38Z", "mode": "train", "global_step": 3258, "epoch": 0.32727272727272727, "loss": 0.008, "grad_norm": 5.533945083618164, "learning_rate": 1.3030303030303033e-07, "num_tokens": 6114883.0, "completions/mean_length": 101.625, "completions/min_length": 91.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.625, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.9797216653823853, "rewards/meter/std": 0.02639707550406456, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8109403848648071, "rewards/repeat_soft/std": 0.029773596674203873, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7979687452316284, "rewards/total_composite/std": 0.011949531733989716, "reward": 0.7979687452316284, "reward_std": 0.011949542909860611, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07683674991130829, "sampling/sampling_logp_difference/max": 1.6352424621582031, "sampling/importance_sampling_ratio/min": 0.1949051022529602, "sampling/importance_sampling_ratio/mean": 1.0069694519042969, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.32338233664631844, "clip_ratio/low_mean": 0.020340909250080585, "clip_ratio/low_min": 0.020340909250080585, "clip_ratio/high_mean": 0.06795409927144647, "clip_ratio/high_max": 0.06795409927144647, "clip_ratio/region_mean": 0.08829500852152705, "reward_total_mean": 0.7979687452316284, "reward_meter_mean": 0.9797216653823853, "reward_meter_std": 0.02639707550406456, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8109403848648071, "reward_repeat_soft_std": 0.029773596674203873, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7979687452316284, "reward_total_composite_std": 0.011949531733989716} {"timestamp_utc": "2026-04-13T06:29:45Z", "mode": "train", "global_step": 3259, "epoch": 0.327373179306881, "loss": 0.0121, "grad_norm": 7.019510269165039, "learning_rate": 1.272727272727273e-07, "num_tokens": 6117244.0, "completions/mean_length": 122.125, "completions/min_length": 111.0, "completions/max_length": 137.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.125, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.7189546227455139, "rewards/meter/std": 0.24160712957382202, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8846191167831421, "rewards/repeat_soft/std": 0.0637100487947464, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.6624914407730103, "rewards/total_composite/std": 0.11300024390220642, "reward": 0.6624914407730103, "reward_std": 0.11300022900104523, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10092660784721375, "sampling/sampling_logp_difference/max": 2.305201768875122, "sampling/importance_sampling_ratio/min": 0.09973867237567902, "sampling/importance_sampling_ratio/mean": 1.0030452013015747, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5246684290468693, "clip_ratio/low_mean": 0.03622628375887871, "clip_ratio/low_min": 0.03622628375887871, "clip_ratio/high_mean": 0.06997611187398434, "clip_ratio/high_max": 0.06997611187398434, "clip_ratio/region_mean": 0.10620239563286304, "reward_total_mean": 0.6624914407730103, "reward_meter_mean": 0.7189546227455139, "reward_meter_std": 0.24160712957382202, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8846191167831421, "reward_repeat_soft_std": 0.0637100487947464, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.6624914407730103, "reward_total_composite_std": 0.11300024390220642} {"timestamp_utc": "2026-04-13T06:29:52Z", "mode": "train", "global_step": 3260, "epoch": 0.32747363134103463, "loss": -0.0116, "grad_norm": 6.3011088371276855, "learning_rate": 1.2424242424242426e-07, "num_tokens": 6119214.0, "completions/mean_length": 87.25, "completions/min_length": 80.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.25, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.970553457736969, "rewards/meter/std": 0.037305884063243866, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9564842581748962, "rewards/repeat_soft/std": 0.03045623190701008, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.8207725286483765, "rewards/total_composite/std": 0.06969308853149414, "reward": 0.8207725286483765, "reward_std": 0.06969308853149414, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09080459177494049, "sampling/sampling_logp_difference/max": 1.7695503234863281, "sampling/importance_sampling_ratio/min": 0.17040960490703583, "sampling/importance_sampling_ratio/mean": 0.999954104423523, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47964635863900185, "clip_ratio/low_mean": 0.05644855182617903, "clip_ratio/low_min": 0.05644855182617903, "clip_ratio/high_mean": 0.014270887710154057, "clip_ratio/high_max": 0.014270887710154057, "clip_ratio/region_mean": 0.07071943953633308, "reward_total_mean": 0.8207725286483765, "reward_meter_mean": 0.970553457736969, "reward_meter_std": 0.037305884063243866, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9564842581748962, "reward_repeat_soft_std": 0.03045623190701008, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.8207725286483765, "reward_total_composite_std": 0.06969308853149414} {"timestamp_utc": "2026-04-13T06:29:59Z", "mode": "train", "global_step": 3261, "epoch": 0.32757408337518834, "loss": 0.0173, "grad_norm": 5.5211358070373535, "learning_rate": 1.2121212121212122e-07, "num_tokens": 6121287.0, "completions/mean_length": 93.125, "completions/min_length": 87.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.125, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9927302598953247, "rewards/meter/std": 0.0027025712188333273, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9177069067955017, "rewards/repeat_soft/std": 0.050144653767347336, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.0843462198972702, "rewards/total_composite/mean": 0.8039993047714233, "rewards/total_composite/std": 0.02293190360069275, "reward": 0.8039993047714233, "reward_std": 0.0229319017380476, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08278486132621765, "sampling/sampling_logp_difference/max": 2.079502582550049, "sampling/importance_sampling_ratio/min": 0.12499237060546875, "sampling/importance_sampling_ratio/mean": 1.0011858940124512, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37834257259964943, "clip_ratio/low_mean": 0.020263557322323322, "clip_ratio/low_min": 0.020263557322323322, "clip_ratio/high_mean": 0.06136521929875016, "clip_ratio/high_max": 0.06136521929875016, "clip_ratio/region_mean": 0.08162877662107348, "reward_total_mean": 0.8039993047714233, "reward_meter_mean": 0.9927302598953247, "reward_meter_std": 0.0027025712188333273, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9177069067955017, "reward_repeat_soft_std": 0.050144653767347336, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.0843462198972702, "reward_total_composite_mean": 0.8039993047714233, "reward_total_composite_std": 0.02293190360069275} {"timestamp_utc": "2026-04-13T06:30:05Z", "mode": "train", "global_step": 3262, "epoch": 0.32767453540934205, "loss": 0.0389, "grad_norm": 9.218621253967285, "learning_rate": 1.1818181818181818e-07, "num_tokens": 6123154.0, "completions/mean_length": 66.375, "completions/min_length": 59.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.375, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9917113780975342, "rewards/meter/std": 0.001778894104063511, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9657808542251587, "rewards/repeat_soft/std": 0.03847542405128479, "rewards/judge_quality/mean": 0.4024999737739563, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.8135982155799866, "rewards/total_composite/std": 0.019675401970744133, "reward": 0.8135982155799866, "reward_std": 0.019675400108098984, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08212856203317642, "sampling/sampling_logp_difference/max": 1.0436034202575684, "sampling/importance_sampling_ratio/min": 0.35218334197998047, "sampling/importance_sampling_ratio/mean": 1.0034244060516357, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.41286228597164154, "clip_ratio/low_mean": 0.013010540511459112, "clip_ratio/low_min": 0.013010540511459112, "clip_ratio/high_mean": 0.05896029528230429, "clip_ratio/high_max": 0.05896029528230429, "clip_ratio/region_mean": 0.0719708357937634, "reward_total_mean": 0.8135982155799866, "reward_meter_mean": 0.9917113780975342, "reward_meter_std": 0.001778894104063511, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9657808542251587, "reward_repeat_soft_std": 0.03847542405128479, "reward_judge_quality_mean": 0.4024999737739563, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.8135982155799866, "reward_total_composite_std": 0.019675401970744133} {"timestamp_utc": "2026-04-13T06:30:12Z", "mode": "train", "global_step": 3263, "epoch": 0.32777498744349576, "loss": 0.0348, "grad_norm": 5.428579330444336, "learning_rate": 1.1515151515151516e-07, "num_tokens": 6125787.0, "completions/mean_length": 144.125, "completions/min_length": 136.0, "completions/max_length": 154.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 144.125, "completions/min_terminated_length": 136.0, "completions/max_terminated_length": 154.0, "rewards/meter/mean": 0.8509034514427185, "rewards/meter/std": 0.20843879878520966, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8861290812492371, "rewards/repeat_soft/std": 0.055426497012376785, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7475194334983826, "rewards/total_composite/std": 0.09385190159082413, "reward": 0.7475194334983826, "reward_std": 0.09385190159082413, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0600588321685791, "sampling/sampling_logp_difference/max": 2.1322097778320312, "sampling/importance_sampling_ratio/min": 0.11857497692108154, "sampling/importance_sampling_ratio/mean": 1.0048571825027466, "sampling/importance_sampling_ratio/max": 1.9365289211273193, "entropy": 0.30855587869882584, "clip_ratio/low_mean": 0.009941782336682081, "clip_ratio/low_min": 0.009941782336682081, "clip_ratio/high_mean": 0.04028267040848732, "clip_ratio/high_max": 0.04028267040848732, "clip_ratio/region_mean": 0.0502244527451694, "reward_total_mean": 0.7475194334983826, "reward_meter_mean": 0.8509034514427185, "reward_meter_std": 0.20843879878520966, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8861290812492371, "reward_repeat_soft_std": 0.055426497012376785, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7475194334983826, "reward_total_composite_std": 0.09385190159082413} {"timestamp_utc": "2026-04-13T06:30:18Z", "mode": "train", "global_step": 3264, "epoch": 0.3278754394776494, "loss": -0.004, "grad_norm": 12.335519790649414, "learning_rate": 1.1212121212121213e-07, "num_tokens": 6127246.0, "completions/mean_length": 24.375, "completions/min_length": 21.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.375, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.5897924900054932, "rewards/meter/std": 0.39950984716415405, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9534652829170227, "rewards/repeat_soft/std": 0.018588263541460037, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.6348781585693359, "rewards/total_composite/std": 0.18222424387931824, "reward": 0.6348781585693359, "reward_std": 0.18222424387931824, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12066757678985596, "sampling/sampling_logp_difference/max": 1.3248625993728638, "sampling/importance_sampling_ratio/min": 0.26583948731422424, "sampling/importance_sampling_ratio/mean": 1.0063538551330566, "sampling/importance_sampling_ratio/max": 1.7823179960250854, "entropy": 0.6641268581151962, "clip_ratio/low_mean": 0.07459627464413643, "clip_ratio/low_min": 0.07459627464413643, "clip_ratio/high_mean": 0.09329212550073862, "clip_ratio/high_max": 0.09329212550073862, "clip_ratio/region_mean": 0.16788840014487505, "reward_total_mean": 0.6348781585693359, "reward_meter_mean": 0.5897924900054932, "reward_meter_std": 0.39950984716415405, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9534652829170227, "reward_repeat_soft_std": 0.018588263541460037, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.6348781585693359, "reward_total_composite_std": 0.18222424387931824} {"timestamp_utc": "2026-04-13T06:30:25Z", "mode": "train", "global_step": 3265, "epoch": 0.3279758915118031, "loss": -0.0237, "grad_norm": 7.0089616775512695, "learning_rate": 1.090909090909091e-07, "num_tokens": 6129155.0, "completions/mean_length": 69.625, "completions/min_length": 63.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.625, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9905694723129272, "rewards/meter/std": 0.00745804188773036, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9821208119392395, "rewards/repeat_soft/std": 0.013059220276772976, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.876218318939209, "rewards/total_composite/std": 0.07672913372516632, "reward": 0.876218318939209, "reward_std": 0.07672914862632751, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07559535652399063, "sampling/sampling_logp_difference/max": 1.0756173133850098, "sampling/importance_sampling_ratio/min": 0.34108713269233704, "sampling/importance_sampling_ratio/mean": 1.0145193338394165, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43518565967679024, "clip_ratio/low_mean": 0.04064469737932086, "clip_ratio/low_min": 0.04064469737932086, "clip_ratio/high_mean": 0.03034872282296419, "clip_ratio/high_max": 0.03034872282296419, "clip_ratio/region_mean": 0.07099342020228505, "reward_total_mean": 0.876218318939209, "reward_meter_mean": 0.9905694723129272, "reward_meter_std": 0.00745804188773036, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9821208119392395, "reward_repeat_soft_std": 0.013059220276772976, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.876218318939209, "reward_total_composite_std": 0.07672913372516632} {"timestamp_utc": "2026-04-13T06:30:31Z", "mode": "train", "global_step": 3266, "epoch": 0.32807634354595683, "loss": 0.0094, "grad_norm": 8.71560287475586, "learning_rate": 1.0606060606060608e-07, "num_tokens": 6130894.0, "completions/mean_length": 60.375, "completions/min_length": 56.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.375, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9726320505142212, "rewards/meter/std": 0.063791923224926, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9712517261505127, "rewards/repeat_soft/std": 0.012617296539247036, "rewards/judge_quality/mean": 0.6100000143051147, "rewards/judge_quality/std": 0.20311856269836426, "rewards/total_composite/mean": 0.8678096532821655, "rewards/total_composite/std": 0.07579614222049713, "reward": 0.8678096532821655, "reward_std": 0.07579616457223892, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07604403048753738, "sampling/sampling_logp_difference/max": 1.4068613052368164, "sampling/importance_sampling_ratio/min": 0.24491079151630402, "sampling/importance_sampling_ratio/mean": 0.9959734082221985, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3876117579638958, "clip_ratio/low_mean": 0.03201722167432308, "clip_ratio/low_min": 0.03201722167432308, "clip_ratio/high_mean": 0.04378753900527954, "clip_ratio/high_max": 0.04378753900527954, "clip_ratio/region_mean": 0.07580476067960262, "reward_total_mean": 0.8678096532821655, "reward_meter_mean": 0.9726320505142212, "reward_meter_std": 0.063791923224926, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9712517261505127, "reward_repeat_soft_std": 0.012617296539247036, "reward_judge_quality_mean": 0.6100000143051147, "reward_judge_quality_std": 0.20311856269836426, "reward_total_composite_mean": 0.8678096532821655, "reward_total_composite_std": 0.07579614222049713} {"timestamp_utc": "2026-04-13T06:30:42Z", "mode": "train", "global_step": 3267, "epoch": 0.3281767955801105, "loss": -0.1655, "grad_norm": 1.8540056943893433, "learning_rate": 1.0303030303030304e-07, "num_tokens": 6133015.0, "completions/mean_length": 123.125, "completions/min_length": 66.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 67.5714340209961, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.8488017320632935, "rewards/meter/std": 0.34326595067977905, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.981346845626831, "rewards/repeat_soft/std": 0.01694892719388008, "rewards/judge_quality/mean": 0.3499999940395355, "rewards/judge_quality/std": 0.15445756912231445, "rewards/total_composite/mean": 0.7043141722679138, "rewards/total_composite/std": 0.28509706258773804, "reward": 0.7043141722679138, "reward_std": 0.28509706258773804, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10448502749204636, "sampling/sampling_logp_difference/max": 1.1563529968261719, "sampling/importance_sampling_ratio/min": 0.31463155150413513, "sampling/importance_sampling_ratio/mean": 1.006600022315979, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6138860508799553, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10031449841335416, "clip_ratio/high_max": 0.10031449841335416, "clip_ratio/region_mean": 0.10031449841335416, "reward_total_mean": 0.7043141722679138, "reward_meter_mean": 0.8488017320632935, "reward_meter_std": 0.34326595067977905, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.981346845626831, "reward_repeat_soft_std": 0.01694892719388008, "reward_judge_quality_mean": 0.3499999940395355, "reward_judge_quality_std": 0.15445756912231445, "reward_total_composite_mean": 0.7043141722679138, "reward_total_composite_std": 0.28509706258773804} {"timestamp_utc": "2026-04-13T06:30:48Z", "mode": "train", "global_step": 3268, "epoch": 0.3282772476142642, "loss": 0.0203, "grad_norm": 29.4731388092041, "learning_rate": 1.0000000000000001e-07, "num_tokens": 6134473.0, "completions/mean_length": 31.25, "completions/min_length": 28.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.25, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9658690690994263, "rewards/meter/std": 0.050928860902786255, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9482031464576721, "rewards/repeat_soft/std": 0.022486040368676186, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8110864162445068, "rewards/total_composite/std": 0.022149279713630676, "reward": 0.8110864162445068, "reward_std": 0.02214926853775978, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13227182626724243, "sampling/sampling_logp_difference/max": 3.3568594455718994, "sampling/importance_sampling_ratio/min": 0.03484451770782471, "sampling/importance_sampling_ratio/mean": 0.9876549243927002, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5465758591890335, "clip_ratio/low_mean": 0.015196078922599554, "clip_ratio/low_min": 0.015196078922599554, "clip_ratio/high_mean": 0.10477977711707354, "clip_ratio/high_max": 0.10477977711707354, "clip_ratio/region_mean": 0.11997585603967309, "reward_total_mean": 0.8110864162445068, "reward_meter_mean": 0.9658690690994263, "reward_meter_std": 0.050928860902786255, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9482031464576721, "reward_repeat_soft_std": 0.022486040368676186, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8110864162445068, "reward_total_composite_std": 0.022149279713630676} {"timestamp_utc": "2026-04-13T06:31:00Z", "mode": "train", "global_step": 3269, "epoch": 0.3283776996484179, "loss": -0.148, "grad_norm": 1.380052089691162, "learning_rate": 9.696969696969697e-08, "num_tokens": 6136152.0, "completions/mean_length": 115.875, "completions/min_length": 51.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 59.28571701049805, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.891396701335907, "rewards/meter/std": 0.25105416774749756, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9821631908416748, "rewards/repeat_soft/std": 0.00925722811371088, "rewards/judge_quality/mean": 0.4012500047683716, "rewards/judge_quality/std": 0.25176164507865906, "rewards/total_composite/mean": 0.7212857007980347, "rewards/total_composite/std": 0.2984462380409241, "reward": 0.7212857007980347, "reward_std": 0.2984462082386017, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09284783899784088, "sampling/sampling_logp_difference/max": 1.8548027276992798, "sampling/importance_sampling_ratio/min": 0.15648381412029266, "sampling/importance_sampling_ratio/mean": 1.0094151496887207, "sampling/importance_sampling_ratio/max": 1.885738730430603, "entropy": 0.43013090267777443, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06788239860907197, "clip_ratio/high_max": 0.06788239860907197, "clip_ratio/region_mean": 0.06788239860907197, "reward_total_mean": 0.7212857007980347, "reward_meter_mean": 0.891396701335907, "reward_meter_std": 0.25105416774749756, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9821631908416748, "reward_repeat_soft_std": 0.00925722811371088, "reward_judge_quality_mean": 0.4012500047683716, "reward_judge_quality_std": 0.25176164507865906, "reward_total_composite_mean": 0.7212857007980347, "reward_total_composite_std": 0.2984462380409241} {"timestamp_utc": "2026-04-13T06:31:07Z", "mode": "train", "global_step": 3270, "epoch": 0.32847815168257155, "loss": 0.0175, "grad_norm": 11.850172996520996, "learning_rate": 9.393939393939395e-08, "num_tokens": 6137981.0, "completions/mean_length": 68.625, "completions/min_length": 63.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.625, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9851017594337463, "rewards/meter/std": 0.0016707433387637138, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9479454755783081, "rewards/repeat_soft/std": 0.02291940525174141, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.8077153563499451, "rewards/total_composite/std": 0.019788678735494614, "reward": 0.8077153563499451, "reward_std": 0.019788680598139763, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05777110159397125, "sampling/sampling_logp_difference/max": 1.2728819847106934, "sampling/importance_sampling_ratio/min": 0.2800234258174896, "sampling/importance_sampling_ratio/mean": 1.0217829942703247, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3193671554327011, "clip_ratio/low_mean": 0.005514706019312143, "clip_ratio/low_min": 0.005514706019312143, "clip_ratio/high_mean": 0.059114910662174225, "clip_ratio/high_max": 0.059114910662174225, "clip_ratio/region_mean": 0.06462961668148637, "reward_total_mean": 0.8077153563499451, "reward_meter_mean": 0.9851017594337463, "reward_meter_std": 0.0016707433387637138, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9479454755783081, "reward_repeat_soft_std": 0.02291940525174141, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.8077153563499451, "reward_total_composite_std": 0.019788678735494614} {"timestamp_utc": "2026-04-13T06:31:14Z", "mode": "train", "global_step": 3271, "epoch": 0.32857860371672526, "loss": 0.046, "grad_norm": 5.639202117919922, "learning_rate": 9.090909090909091e-08, "num_tokens": 6140411.0, "completions/mean_length": 117.75, "completions/min_length": 109.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.75, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.9859633445739746, "rewards/meter/std": 0.0034315555822104216, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9246061444282532, "rewards/repeat_soft/std": 0.037660639733076096, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.7892690896987915, "rewards/total_composite/std": 0.030406584963202477, "reward": 0.7892690896987915, "reward_std": 0.030406581237912178, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07780894637107849, "sampling/sampling_logp_difference/max": 1.1492588520050049, "sampling/importance_sampling_ratio/min": 0.3168715238571167, "sampling/importance_sampling_ratio/mean": 1.0033783912658691, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40406057238578796, "clip_ratio/low_mean": 0.031230769585818052, "clip_ratio/low_min": 0.031230769585818052, "clip_ratio/high_mean": 0.054213872179389, "clip_ratio/high_max": 0.054213872179389, "clip_ratio/region_mean": 0.08544464176520705, "reward_total_mean": 0.7892690896987915, "reward_meter_mean": 0.9859633445739746, "reward_meter_std": 0.0034315555822104216, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9246061444282532, "reward_repeat_soft_std": 0.037660639733076096, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.7892690896987915, "reward_total_composite_std": 0.030406584963202477} {"timestamp_utc": "2026-04-13T06:31:20Z", "mode": "train", "global_step": 3272, "epoch": 0.32867905575087897, "loss": -0.0292, "grad_norm": 8.47949504852295, "learning_rate": 8.787878787878788e-08, "num_tokens": 6142276.0, "completions/mean_length": 67.125, "completions/min_length": 62.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.125, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9818783402442932, "rewards/meter/std": 0.010988278314471245, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9408926963806152, "rewards/repeat_soft/std": 0.012668561190366745, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.8243095278739929, "rewards/total_composite/std": 0.0299817007035017, "reward": 0.8243095278739929, "reward_std": 0.02998168393969536, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09150989353656769, "sampling/sampling_logp_difference/max": 1.4543781280517578, "sampling/importance_sampling_ratio/min": 0.24792522192001343, "sampling/importance_sampling_ratio/mean": 1.0059534311294556, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4334532469511032, "clip_ratio/low_mean": 0.06663627945818007, "clip_ratio/low_min": 0.06663627945818007, "clip_ratio/high_mean": 0.009740259498357773, "clip_ratio/high_max": 0.009740259498357773, "clip_ratio/region_mean": 0.07637653895653784, "reward_total_mean": 0.8243095278739929, "reward_meter_mean": 0.9818783402442932, "reward_meter_std": 0.010988278314471245, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9408926963806152, "reward_repeat_soft_std": 0.012668561190366745, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.8243095278739929, "reward_total_composite_std": 0.0299817007035017} {"timestamp_utc": "2026-04-13T06:31:27Z", "mode": "train", "global_step": 3273, "epoch": 0.3287795077850326, "loss": 0.044, "grad_norm": 8.472823143005371, "learning_rate": 8.484848484848487e-08, "num_tokens": 6144279.0, "completions/mean_length": 68.375, "completions/min_length": 64.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.375, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.666965126991272, "rewards/meter/std": 0.18714305758476257, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9276244640350342, "rewards/repeat_soft/std": 0.03143106773495674, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.6523967385292053, "rewards/total_composite/std": 0.11074184626340866, "reward": 0.6523967385292053, "reward_std": 0.11074183136224747, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0820414200425148, "sampling/sampling_logp_difference/max": 1.656534194946289, "sampling/importance_sampling_ratio/min": 0.19079910218715668, "sampling/importance_sampling_ratio/mean": 0.9914880394935608, "sampling/importance_sampling_ratio/max": 1.8907365798950195, "entropy": 0.350655697286129, "clip_ratio/low_mean": 0.030031505040824413, "clip_ratio/low_min": 0.030031505040824413, "clip_ratio/high_mean": 0.050487922970205545, "clip_ratio/high_max": 0.050487922970205545, "clip_ratio/region_mean": 0.08051942801102996, "reward_total_mean": 0.6523967385292053, "reward_meter_mean": 0.666965126991272, "reward_meter_std": 0.18714305758476257, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9276244640350342, "reward_repeat_soft_std": 0.03143106773495674, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.6523967385292053, "reward_total_composite_std": 0.11074184626340866} {"timestamp_utc": "2026-04-13T06:31:38Z", "mode": "train", "global_step": 3274, "epoch": 0.32887995981918633, "loss": -0.2463, "grad_norm": 1.3652381896972656, "learning_rate": 8.181818181818183e-08, "num_tokens": 6147112.0, "completions/mean_length": 220.125, "completions/min_length": 173.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 178.42857360839844, "completions/min_terminated_length": 173.0, "completions/max_terminated_length": 186.0, "rewards/meter/mean": 0.9910409450531006, "rewards/meter/std": 0.005688815377652645, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8229356408119202, "rewards/repeat_soft/std": 0.0619242824614048, "rewards/judge_quality/mean": 0.20875000953674316, "rewards/judge_quality/std": 0.11038083583116531, "rewards/total_composite/mean": 0.6525471806526184, "rewards/total_composite/std": 0.2653796970844269, "reward": 0.6525471806526184, "reward_std": 0.2653796970844269, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06291408836841583, "sampling/sampling_logp_difference/max": 1.764817714691162, "sampling/importance_sampling_ratio/min": 0.17121799290180206, "sampling/importance_sampling_ratio/mean": 1.008528470993042, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2944665774703026, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.05887332232668996, "clip_ratio/high_max": 0.05887332232668996, "clip_ratio/region_mean": 0.05887332232668996, "reward_total_mean": 0.6525471806526184, "reward_meter_mean": 0.9910409450531006, "reward_meter_std": 0.005688815377652645, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8229356408119202, "reward_repeat_soft_std": 0.0619242824614048, "reward_judge_quality_mean": 0.20875000953674316, "reward_judge_quality_std": 0.11038083583116531, "reward_total_composite_mean": 0.6525471806526184, "reward_total_composite_std": 0.2653796970844269} {"timestamp_utc": "2026-04-13T06:31:46Z", "mode": "train", "global_step": 3275, "epoch": 0.32898041185334004, "loss": 0.0123, "grad_norm": 4.71132755279541, "learning_rate": 7.87878787878788e-08, "num_tokens": 6150090.0, "completions/mean_length": 178.25, "completions/min_length": 174.0, "completions/max_length": 182.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 178.25, "completions/min_terminated_length": 174.0, "completions/max_terminated_length": 182.0, "rewards/meter/mean": 0.984290361404419, "rewards/meter/std": 0.016317861154675484, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7254549860954285, "rewards/repeat_soft/std": 0.07427062839269638, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7787261009216309, "rewards/total_composite/std": 0.024270573630928993, "reward": 0.7787261009216309, "reward_std": 0.02427058108150959, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06374892592430115, "sampling/sampling_logp_difference/max": 1.6879558563232422, "sampling/importance_sampling_ratio/min": 0.18489709496498108, "sampling/importance_sampling_ratio/mean": 1.0093430280685425, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2970374785363674, "clip_ratio/low_mean": 0.02164137503132224, "clip_ratio/low_min": 0.02164137503132224, "clip_ratio/high_mean": 0.037911586463451385, "clip_ratio/high_max": 0.037911586463451385, "clip_ratio/region_mean": 0.059552961494773626, "reward_total_mean": 0.7787261009216309, "reward_meter_mean": 0.984290361404419, "reward_meter_std": 0.016317861154675484, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7254549860954285, "reward_repeat_soft_std": 0.07427062839269638, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7787261009216309, "reward_total_composite_std": 0.024270573630928993} {"timestamp_utc": "2026-04-13T06:31:53Z", "mode": "train", "global_step": 3276, "epoch": 0.32908086388749375, "loss": 0.0345, "grad_norm": 7.756103992462158, "learning_rate": 7.575757575757576e-08, "num_tokens": 6151921.0, "completions/mean_length": 59.875, "completions/min_length": 56.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.875, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.8448272943496704, "rewards/meter/std": 0.2563541829586029, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9507598280906677, "rewards/repeat_soft/std": 0.04275740683078766, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.8284982442855835, "rewards/total_composite/std": 0.1324203759431839, "reward": 0.8284982442855835, "reward_std": 0.1324203759431839, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07213463634252548, "sampling/sampling_logp_difference/max": 1.3398971557617188, "sampling/importance_sampling_ratio/min": 0.2618725895881653, "sampling/importance_sampling_ratio/mean": 0.9978637099266052, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37858573719859123, "clip_ratio/low_mean": 0.04926764313131571, "clip_ratio/low_min": 0.04926764313131571, "clip_ratio/high_mean": 0.016886098077520728, "clip_ratio/high_max": 0.016886098077520728, "clip_ratio/region_mean": 0.06615374120883644, "reward_total_mean": 0.8284982442855835, "reward_meter_mean": 0.8448272943496704, "reward_meter_std": 0.2563541829586029, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9507598280906677, "reward_repeat_soft_std": 0.04275740683078766, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.8284982442855835, "reward_total_composite_std": 0.1324203759431839} {"timestamp_utc": "2026-04-13T06:31:59Z", "mode": "train", "global_step": 3277, "epoch": 0.3291813159216474, "loss": 0.0234, "grad_norm": 7.84608793258667, "learning_rate": 7.272727272727274e-08, "num_tokens": 6153519.0, "completions/mean_length": 57.75, "completions/min_length": 54.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.75, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9855232238769531, "rewards/meter/std": 0.004798950161784887, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9633569717407227, "rewards/repeat_soft/std": 0.026658331975340843, "rewards/judge_quality/mean": 0.5087499618530273, "rewards/judge_quality/std": 0.16617010533809662, "rewards/total_composite/mean": 0.8424460887908936, "rewards/total_composite/std": 0.04851274937391281, "reward": 0.8424460887908936, "reward_std": 0.048512741923332214, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09133017063140869, "sampling/sampling_logp_difference/max": 3.763878345489502, "sampling/importance_sampling_ratio/min": 0.023193612694740295, "sampling/importance_sampling_ratio/mean": 1.0150395631790161, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40032321587204933, "clip_ratio/low_mean": 0.07128078769892454, "clip_ratio/low_min": 0.07128078769892454, "clip_ratio/high_mean": 0.01315789483487606, "clip_ratio/high_max": 0.01315789483487606, "clip_ratio/region_mean": 0.0844386825338006, "reward_total_mean": 0.8424460887908936, "reward_meter_mean": 0.9855232238769531, "reward_meter_std": 0.004798950161784887, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9633569717407227, "reward_repeat_soft_std": 0.026658331975340843, "reward_judge_quality_mean": 0.5087499618530273, "reward_judge_quality_std": 0.16617010533809662, "reward_total_composite_mean": 0.8424460887908936, "reward_total_composite_std": 0.04851274937391281} {"timestamp_utc": "2026-04-13T06:32:06Z", "mode": "train", "global_step": 3278, "epoch": 0.3292817679558011, "loss": -0.0191, "grad_norm": 6.280437469482422, "learning_rate": 6.96969696969697e-08, "num_tokens": 6156232.0, "completions/mean_length": 133.125, "completions/min_length": 118.0, "completions/max_length": 149.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 133.125, "completions/min_terminated_length": 118.0, "completions/max_terminated_length": 149.0, "rewards/meter/mean": 0.6774550080299377, "rewards/meter/std": 0.42104536294937134, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9248989224433899, "rewards/repeat_soft/std": 0.08098739385604858, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.673344612121582, "rewards/total_composite/std": 0.18885163962841034, "reward": 0.673344612121582, "reward_std": 0.18885163962841034, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09091738611459732, "sampling/sampling_logp_difference/max": 3.073216438293457, "sampling/importance_sampling_ratio/min": 0.04627208411693573, "sampling/importance_sampling_ratio/mean": 0.9921719431877136, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31676895171403885, "clip_ratio/low_mean": 0.029677086509764194, "clip_ratio/low_min": 0.029677086509764194, "clip_ratio/high_mean": 0.06900172308087349, "clip_ratio/high_max": 0.06900172308087349, "clip_ratio/region_mean": 0.09867880959063768, "reward_total_mean": 0.673344612121582, "reward_meter_mean": 0.6774550080299377, "reward_meter_std": 0.42104536294937134, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9248989224433899, "reward_repeat_soft_std": 0.08098739385604858, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.673344612121582, "reward_total_composite_std": 0.18885163962841034} {"timestamp_utc": "2026-04-13T06:32:13Z", "mode": "train", "global_step": 3279, "epoch": 0.3293822199899548, "loss": 0.0367, "grad_norm": 7.6552510261535645, "learning_rate": 6.666666666666668e-08, "num_tokens": 6158594.0, "completions/mean_length": 113.25, "completions/min_length": 108.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.25, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.9937226176261902, "rewards/meter/std": 0.0014749052934348583, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9145458936691284, "rewards/repeat_soft/std": 0.036692265421152115, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.8018797636032104, "rewards/total_composite/std": 0.02567218989133835, "reward": 0.8018797636032104, "reward_std": 0.02567218989133835, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08678615093231201, "sampling/sampling_logp_difference/max": 1.5700531005859375, "sampling/importance_sampling_ratio/min": 0.20803412795066833, "sampling/importance_sampling_ratio/mean": 0.9984867572784424, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5088436789810658, "clip_ratio/low_mean": 0.015293668955564499, "clip_ratio/low_min": 0.015293668955564499, "clip_ratio/high_mean": 0.06974570173770189, "clip_ratio/high_max": 0.06974570173770189, "clip_ratio/region_mean": 0.08503937069326639, "reward_total_mean": 0.8018797636032104, "reward_meter_mean": 0.9937226176261902, "reward_meter_std": 0.0014749052934348583, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9145458936691284, "reward_repeat_soft_std": 0.036692265421152115, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.8018797636032104, "reward_total_composite_std": 0.02567218989133835} {"timestamp_utc": "2026-04-13T06:32:21Z", "mode": "train", "global_step": 3280, "epoch": 0.32948267202410847, "loss": 0.0132, "grad_norm": 5.295575141906738, "learning_rate": 6.363636363636365e-08, "num_tokens": 6161337.0, "completions/mean_length": 140.875, "completions/min_length": 132.0, "completions/max_length": 151.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 140.875, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 151.0, "rewards/meter/mean": 0.9928926825523376, "rewards/meter/std": 0.004879934247583151, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9414960145950317, "rewards/repeat_soft/std": 0.016241740435361862, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.7830138206481934, "rewards/total_composite/std": 0.03843383118510246, "reward": 0.7830138206481934, "reward_std": 0.03843381628394127, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08121262490749359, "sampling/sampling_logp_difference/max": 1.6902761459350586, "sampling/importance_sampling_ratio/min": 0.1844685673713684, "sampling/importance_sampling_ratio/mean": 1.0081065893173218, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48518020659685135, "clip_ratio/low_mean": 0.04109468264505267, "clip_ratio/low_min": 0.04109468264505267, "clip_ratio/high_mean": 0.04043578915297985, "clip_ratio/high_max": 0.04043578915297985, "clip_ratio/region_mean": 0.08153047179803252, "reward_total_mean": 0.7830138206481934, "reward_meter_mean": 0.9928926825523376, "reward_meter_std": 0.004879934247583151, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9414960145950317, "reward_repeat_soft_std": 0.016241740435361862, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.7830138206481934, "reward_total_composite_std": 0.03843383118510246} {"timestamp_utc": "2026-04-13T06:32:28Z", "mode": "train", "global_step": 3281, "epoch": 0.3295831240582622, "loss": 0.0407, "grad_norm": 7.539600372314453, "learning_rate": 6.060606060606061e-08, "num_tokens": 6163221.0, "completions/mean_length": 81.5, "completions/min_length": 75.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.5, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9863647222518921, "rewards/meter/std": 0.019480455666780472, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9246857166290283, "rewards/repeat_soft/std": 0.03814719244837761, "rewards/judge_quality/mean": 0.44875001907348633, "rewards/judge_quality/std": 0.21256513893604279, "rewards/total_composite/mean": 0.8209577202796936, "rewards/total_composite/std": 0.06357122212648392, "reward": 0.8209577202796936, "reward_std": 0.06357122212648392, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10423505306243896, "sampling/sampling_logp_difference/max": 1.2479121685028076, "sampling/importance_sampling_ratio/min": 0.2871035933494568, "sampling/importance_sampling_ratio/mean": 1.0027377605438232, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49532705545425415, "clip_ratio/low_mean": 0.07472571777179837, "clip_ratio/low_min": 0.07472571777179837, "clip_ratio/high_mean": 0.014999999664723873, "clip_ratio/high_max": 0.014999999664723873, "clip_ratio/region_mean": 0.08972571743652225, "reward_total_mean": 0.8209577202796936, "reward_meter_mean": 0.9863647222518921, "reward_meter_std": 0.019480455666780472, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9246857166290283, "reward_repeat_soft_std": 0.03814719244837761, "reward_judge_quality_mean": 0.44875001907348633, "reward_judge_quality_std": 0.21256513893604279, "reward_total_composite_mean": 0.8209577202796936, "reward_total_composite_std": 0.06357122212648392} {"timestamp_utc": "2026-04-13T06:32:34Z", "mode": "train", "global_step": 3282, "epoch": 0.3296835760924159, "loss": 0.0766, "grad_norm": 9.272500991821289, "learning_rate": 5.757575757575758e-08, "num_tokens": 6164872.0, "completions/mean_length": 53.375, "completions/min_length": 41.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.375, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.7222088575363159, "rewards/meter/std": 0.3035672903060913, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9777908325195312, "rewards/repeat_soft/std": 0.027943924069404602, "rewards/judge_quality/mean": 0.8612500429153442, "rewards/judge_quality/std": 0.16617010533809662, "rewards/total_composite/mean": 0.8311480283737183, "rewards/total_composite/std": 0.13411763310432434, "reward": 0.8311480283737183, "reward_std": 0.13411761820316315, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1018884927034378, "sampling/sampling_logp_difference/max": 1.3886449337005615, "sampling/importance_sampling_ratio/min": 0.24941305816173553, "sampling/importance_sampling_ratio/mean": 0.9961822032928467, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4196528121829033, "clip_ratio/low_mean": 0.043355082627385855, "clip_ratio/low_min": 0.043355082627385855, "clip_ratio/high_mean": 0.052002618089318275, "clip_ratio/high_max": 0.052002618089318275, "clip_ratio/region_mean": 0.09535770071670413, "reward_total_mean": 0.8311480283737183, "reward_meter_mean": 0.7222088575363159, "reward_meter_std": 0.3035672903060913, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9777908325195312, "reward_repeat_soft_std": 0.027943924069404602, "reward_judge_quality_mean": 0.8612500429153442, "reward_judge_quality_std": 0.16617010533809662, "reward_total_composite_mean": 0.8311480283737183, "reward_total_composite_std": 0.13411763310432434} {"timestamp_utc": "2026-04-13T06:32:41Z", "mode": "train", "global_step": 3283, "epoch": 0.32978402812656954, "loss": -0.0009, "grad_norm": 6.80902099609375, "learning_rate": 5.454545454545455e-08, "num_tokens": 6167139.0, "completions/mean_length": 115.375, "completions/min_length": 107.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.375, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.9324328303337097, "rewards/meter/std": 0.15952973067760468, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9271476864814758, "rewards/repeat_soft/std": 0.057270895689725876, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7883095145225525, "rewards/total_composite/std": 0.06986506283283234, "reward": 0.7883095145225525, "reward_std": 0.06986507028341293, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07753002643585205, "sampling/sampling_logp_difference/max": 3.084552049636841, "sampling/importance_sampling_ratio/min": 0.04575052112340927, "sampling/importance_sampling_ratio/mean": 1.0072799921035767, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.383933961391449, "clip_ratio/low_mean": 0.011061946861445904, "clip_ratio/low_min": 0.011061946861445904, "clip_ratio/high_mean": 0.07657495932653546, "clip_ratio/high_max": 0.07657495932653546, "clip_ratio/region_mean": 0.08763690618798137, "reward_total_mean": 0.7883095145225525, "reward_meter_mean": 0.9324328303337097, "reward_meter_std": 0.15952973067760468, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9271476864814758, "reward_repeat_soft_std": 0.057270895689725876, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7883095145225525, "reward_total_composite_std": 0.06986506283283234} {"timestamp_utc": "2026-04-13T06:32:48Z", "mode": "train", "global_step": 3284, "epoch": 0.32988448016072325, "loss": 0.0018, "grad_norm": 6.280606269836426, "learning_rate": 5.151515151515152e-08, "num_tokens": 6169712.0, "completions/mean_length": 137.625, "completions/min_length": 134.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 137.625, "completions/min_terminated_length": 134.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.9923058748245239, "rewards/meter/std": 0.004598211031407118, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9515643119812012, "rewards/repeat_soft/std": 0.042526595294475555, "rewards/judge_quality/mean": 0.5575000047683716, "rewards/judge_quality/std": 0.19955310225486755, "rewards/total_composite/mean": 0.8589440584182739, "rewards/total_composite/std": 0.06065918505191803, "reward": 0.8589440584182739, "reward_std": 0.060659196227788925, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07806216180324554, "sampling/sampling_logp_difference/max": 2.270536422729492, "sampling/importance_sampling_ratio/min": 0.10325677692890167, "sampling/importance_sampling_ratio/mean": 1.0098826885223389, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.41379762440919876, "clip_ratio/low_mean": 0.03886065911501646, "clip_ratio/low_min": 0.03886065911501646, "clip_ratio/high_mean": 0.02301377709954977, "clip_ratio/high_max": 0.02301377709954977, "clip_ratio/region_mean": 0.06187443621456623, "reward_total_mean": 0.8589440584182739, "reward_meter_mean": 0.9923058748245239, "reward_meter_std": 0.004598211031407118, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9515643119812012, "reward_repeat_soft_std": 0.042526595294475555, "reward_judge_quality_mean": 0.5575000047683716, "reward_judge_quality_std": 0.19955310225486755, "reward_total_composite_mean": 0.8589440584182739, "reward_total_composite_std": 0.06065918505191803} {"timestamp_utc": "2026-04-13T06:32:54Z", "mode": "train", "global_step": 3285, "epoch": 0.32998493219487696, "loss": 0.0033, "grad_norm": 9.51995849609375, "learning_rate": 4.8484848484848486e-08, "num_tokens": 6171443.0, "completions/mean_length": 54.375, "completions/min_length": 49.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.375, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9881882667541504, "rewards/meter/std": 0.0041625783778727055, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.981923520565033, "rewards/repeat_soft/std": 0.008813531138002872, "rewards/judge_quality/mean": 0.7737500667572021, "rewards/judge_quality/std": 0.22032040357589722, "rewards/total_composite/mean": 0.9250020980834961, "rewards/total_composite/std": 0.0662836804986, "reward": 0.9250020980834961, "reward_std": 0.0662836879491806, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10289929807186127, "sampling/sampling_logp_difference/max": 2.322277069091797, "sampling/importance_sampling_ratio/min": 0.0980500653386116, "sampling/importance_sampling_ratio/mean": 0.9956657886505127, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.501244068145752, "clip_ratio/low_mean": 0.03744222642853856, "clip_ratio/low_min": 0.03744222642853856, "clip_ratio/high_mean": 0.07755595538765192, "clip_ratio/high_max": 0.07755595538765192, "clip_ratio/region_mean": 0.11499818181619048, "reward_total_mean": 0.9250020980834961, "reward_meter_mean": 0.9881882667541504, "reward_meter_std": 0.0041625783778727055, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.981923520565033, "reward_repeat_soft_std": 0.008813531138002872, "reward_judge_quality_mean": 0.7737500667572021, "reward_judge_quality_std": 0.22032040357589722, "reward_total_composite_mean": 0.9250020980834961, "reward_total_composite_std": 0.0662836804986} {"timestamp_utc": "2026-04-13T06:33:01Z", "mode": "train", "global_step": 3286, "epoch": 0.33008538422903067, "loss": 0.0049, "grad_norm": 7.2535858154296875, "learning_rate": 4.545454545454546e-08, "num_tokens": 6173613.0, "completions/mean_length": 114.25, "completions/min_length": 106.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.25, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.976543128490448, "rewards/meter/std": 0.016373468562960625, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8816501498222351, "rewards/repeat_soft/std": 0.05292386934161186, "rewards/judge_quality/mean": 0.6825000047683716, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.8776719570159912, "rewards/total_composite/std": 0.066607765853405, "reward": 0.8776719570159912, "reward_std": 0.0666077509522438, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06934276968240738, "sampling/sampling_logp_difference/max": 1.502429485321045, "sampling/importance_sampling_ratio/min": 0.22258873283863068, "sampling/importance_sampling_ratio/mean": 1.0001747608184814, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3280685357749462, "clip_ratio/low_mean": 0.030373204965144396, "clip_ratio/low_min": 0.030373204965144396, "clip_ratio/high_mean": 0.03836646629497409, "clip_ratio/high_max": 0.03836646629497409, "clip_ratio/region_mean": 0.06873967126011848, "reward_total_mean": 0.8776719570159912, "reward_meter_mean": 0.976543128490448, "reward_meter_std": 0.016373468562960625, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8816501498222351, "reward_repeat_soft_std": 0.05292386934161186, "reward_judge_quality_mean": 0.6825000047683716, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.8776719570159912, "reward_total_composite_std": 0.066607765853405} {"timestamp_utc": "2026-04-13T06:33:07Z", "mode": "train", "global_step": 3287, "epoch": 0.3301858362631843, "loss": -0.0063, "grad_norm": 10.388537406921387, "learning_rate": 4.2424242424242435e-08, "num_tokens": 6175244.0, "completions/mean_length": 47.875, "completions/min_length": 44.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.875, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.8930122256278992, "rewards/meter/std": 0.1820669025182724, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9711524844169617, "rewards/repeat_soft/std": 0.02188713103532791, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7622207403182983, "rewards/total_composite/std": 0.07777740061283112, "reward": 0.7622207403182983, "reward_std": 0.07777741551399231, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08600202202796936, "sampling/sampling_logp_difference/max": 2.1166534423828125, "sampling/importance_sampling_ratio/min": 0.1204340010881424, "sampling/importance_sampling_ratio/mean": 1.0017828941345215, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48315706476569176, "clip_ratio/low_mean": 0.02676877425983548, "clip_ratio/low_min": 0.02676877425983548, "clip_ratio/high_mean": 0.05124652967788279, "clip_ratio/high_max": 0.05124652967788279, "clip_ratio/region_mean": 0.07801530393771827, "reward_total_mean": 0.7622207403182983, "reward_meter_mean": 0.8930122256278992, "reward_meter_std": 0.1820669025182724, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9711524844169617, "reward_repeat_soft_std": 0.02188713103532791, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7622207403182983, "reward_total_composite_std": 0.07777740061283112} {"timestamp_utc": "2026-04-13T06:33:14Z", "mode": "train", "global_step": 3288, "epoch": 0.330286288297338, "loss": 0.021, "grad_norm": 12.852709770202637, "learning_rate": 3.93939393939394e-08, "num_tokens": 6177038.0, "completions/mean_length": 62.25, "completions/min_length": 55.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.25, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9944112300872803, "rewards/meter/std": 0.0024336182978004217, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9834589958190918, "rewards/repeat_soft/std": 0.009931392967700958, "rewards/judge_quality/mean": 0.7887499928474426, "rewards/judge_quality/std": 0.21918272972106934, "rewards/total_composite/mean": 0.9324559569358826, "rewards/total_composite/std": 0.0648357942700386, "reward": 0.9324559569358826, "reward_std": 0.06483578681945801, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07320044934749603, "sampling/sampling_logp_difference/max": 1.3638120889663696, "sampling/importance_sampling_ratio/min": 0.2556842267513275, "sampling/importance_sampling_ratio/mean": 0.9941254258155823, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3407999910414219, "clip_ratio/low_mean": 0.016129031777381897, "clip_ratio/low_min": 0.016129031777381897, "clip_ratio/high_mean": 0.05581540986895561, "clip_ratio/high_max": 0.05581540986895561, "clip_ratio/region_mean": 0.07194444164633751, "reward_total_mean": 0.9324559569358826, "reward_meter_mean": 0.9944112300872803, "reward_meter_std": 0.0024336182978004217, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9834589958190918, "reward_repeat_soft_std": 0.009931392967700958, "reward_judge_quality_mean": 0.7887499928474426, "reward_judge_quality_std": 0.21918272972106934, "reward_total_composite_mean": 0.9324559569358826, "reward_total_composite_std": 0.0648357942700386} {"timestamp_utc": "2026-04-13T06:33:20Z", "mode": "train", "global_step": 3289, "epoch": 0.33038674033149174, "loss": 0.0453, "grad_norm": 12.104578971862793, "learning_rate": 3.636363636363637e-08, "num_tokens": 6178334.0, "completions/mean_length": 22.0, "completions/min_length": 18.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.0, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.9879798889160156, "rewards/meter/std": 0.003219856880605221, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.959877610206604, "rewards/repeat_soft/std": 0.007417192216962576, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.8222036361694336, "rewards/total_composite/std": 0.004208676051348448, "reward": 0.8222036361694336, "reward_std": 0.004208686761558056, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06078128516674042, "sampling/sampling_logp_difference/max": 1.2002172470092773, "sampling/importance_sampling_ratio/min": 0.3011288046836853, "sampling/importance_sampling_ratio/mean": 1.0035736560821533, "sampling/importance_sampling_ratio/max": 1.4960826635360718, "entropy": 0.3309175428003073, "clip_ratio/low_mean": 0.013888888992369175, "clip_ratio/low_min": 0.013888888992369175, "clip_ratio/high_mean": 0.03937500016763806, "clip_ratio/high_max": 0.03937500016763806, "clip_ratio/region_mean": 0.05326388916000724, "reward_total_mean": 0.8222036361694336, "reward_meter_mean": 0.9879798889160156, "reward_meter_std": 0.003219856880605221, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.959877610206604, "reward_repeat_soft_std": 0.007417192216962576, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.8222036361694336, "reward_total_composite_std": 0.004208676051348448} {"timestamp_utc": "2026-04-13T06:33:27Z", "mode": "train", "global_step": 3290, "epoch": 0.3304871923656454, "loss": 0.0519, "grad_norm": 5.686668872833252, "learning_rate": 3.333333333333334e-08, "num_tokens": 6180645.0, "completions/mean_length": 113.875, "completions/min_length": 102.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.875, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.9836556911468506, "rewards/meter/std": 0.016718611121177673, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8936373591423035, "rewards/repeat_soft/std": 0.04422462359070778, "rewards/judge_quality/mean": 0.5575000047683716, "rewards/judge_quality/std": 0.19955308735370636, "rewards/total_composite/mean": 0.8492587804794312, "rewards/total_composite/std": 0.05641791224479675, "reward": 0.8492587804794312, "reward_std": 0.05641791224479675, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08208899199962616, "sampling/sampling_logp_difference/max": 3.4215474128723145, "sampling/importance_sampling_ratio/min": 0.03266185522079468, "sampling/importance_sampling_ratio/mean": 0.9991361498832703, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36991315707564354, "clip_ratio/low_mean": 0.0387769415974617, "clip_ratio/low_min": 0.0387769415974617, "clip_ratio/high_mean": 0.029437421821057796, "clip_ratio/high_max": 0.029437421821057796, "clip_ratio/region_mean": 0.0682143634185195, "reward_total_mean": 0.8492587804794312, "reward_meter_mean": 0.9836556911468506, "reward_meter_std": 0.016718611121177673, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8936373591423035, "reward_repeat_soft_std": 0.04422462359070778, "reward_judge_quality_mean": 0.5575000047683716, "reward_judge_quality_std": 0.19955308735370636, "reward_total_composite_mean": 0.8492587804794312, "reward_total_composite_std": 0.05641791224479675} {"timestamp_utc": "2026-04-13T06:33:33Z", "mode": "train", "global_step": 3291, "epoch": 0.3305876443997991, "loss": -0.0155, "grad_norm": 9.290125846862793, "learning_rate": 3.0303030303030305e-08, "num_tokens": 6182105.0, "completions/mean_length": 25.5, "completions/min_length": 24.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.5, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9904136657714844, "rewards/meter/std": 0.0011990147177129984, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9392033219337463, "rewards/repeat_soft/std": 0.03512335196137428, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.8542314767837524, "rewards/total_composite/std": 0.06747134774923325, "reward": 0.8542314767837524, "reward_std": 0.06747135519981384, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07431820780038834, "sampling/sampling_logp_difference/max": 1.278723120689392, "sampling/importance_sampling_ratio/min": 0.2783925533294678, "sampling/importance_sampling_ratio/mean": 1.0016961097717285, "sampling/importance_sampling_ratio/max": 1.7832801342010498, "entropy": 0.26981022767722607, "clip_ratio/low_mean": 0.039675925858318806, "clip_ratio/low_min": 0.039675925858318806, "clip_ratio/high_mean": 0.018620689399540424, "clip_ratio/high_max": 0.018620689399540424, "clip_ratio/region_mean": 0.05829661525785923, "reward_total_mean": 0.8542314767837524, "reward_meter_mean": 0.9904136657714844, "reward_meter_std": 0.0011990147177129984, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9392033219337463, "reward_repeat_soft_std": 0.03512335196137428, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.8542314767837524, "reward_total_composite_std": 0.06747134774923325} {"timestamp_utc": "2026-04-13T06:33:41Z", "mode": "train", "global_step": 3292, "epoch": 0.3306880964339528, "loss": -0.0013, "grad_norm": 5.014265060424805, "learning_rate": 2.7272727272727276e-08, "num_tokens": 6184849.0, "completions/mean_length": 138.0, "completions/min_length": 129.0, "completions/max_length": 149.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 138.0, "completions/min_terminated_length": 129.0, "completions/max_terminated_length": 149.0, "rewards/meter/mean": 0.9916599988937378, "rewards/meter/std": 0.0020808640401810408, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7440129518508911, "rewards/repeat_soft/std": 0.033216364681720734, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7838982939720154, "rewards/total_composite/std": 0.023377014324069023, "reward": 0.7838982939720154, "reward_std": 0.02337702549993992, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06854482740163803, "sampling/sampling_logp_difference/max": 2.1212730407714844, "sampling/importance_sampling_ratio/min": 0.11987891793251038, "sampling/importance_sampling_ratio/mean": 1.005699634552002, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2888391949236393, "clip_ratio/low_mean": 0.0065635452046990395, "clip_ratio/low_min": 0.0065635452046990395, "clip_ratio/high_mean": 0.05149642284959555, "clip_ratio/high_max": 0.05149642284959555, "clip_ratio/region_mean": 0.058059968054294586, "reward_total_mean": 0.7838982939720154, "reward_meter_mean": 0.9916599988937378, "reward_meter_std": 0.0020808640401810408, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7440129518508911, "reward_repeat_soft_std": 0.033216364681720734, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7838982939720154, "reward_total_composite_std": 0.023377014324069023} {"timestamp_utc": "2026-04-13T06:33:47Z", "mode": "train", "global_step": 3293, "epoch": 0.33078854846810646, "loss": 0.0342, "grad_norm": 7.179730415344238, "learning_rate": 2.4242424242424243e-08, "num_tokens": 6187038.0, "completions/mean_length": 104.625, "completions/min_length": 93.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.625, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.8931918144226074, "rewards/meter/std": 0.18480555713176727, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9444225430488586, "rewards/repeat_soft/std": 0.034212637692689896, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7596285343170166, "rewards/total_composite/std": 0.08525391668081284, "reward": 0.7596285343170166, "reward_std": 0.08525391668081284, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09811358898878098, "sampling/sampling_logp_difference/max": 2.2777342796325684, "sampling/importance_sampling_ratio/min": 0.10251621901988983, "sampling/importance_sampling_ratio/mean": 1.0033103227615356, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.525936670601368, "clip_ratio/low_mean": 0.013660026248544455, "clip_ratio/low_min": 0.013660026248544455, "clip_ratio/high_mean": 0.07609639596194029, "clip_ratio/high_max": 0.07609639596194029, "clip_ratio/region_mean": 0.08975642221048474, "reward_total_mean": 0.7596285343170166, "reward_meter_mean": 0.8931918144226074, "reward_meter_std": 0.18480555713176727, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9444225430488586, "reward_repeat_soft_std": 0.034212637692689896, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7596285343170166, "reward_total_composite_std": 0.08525391668081284} {"timestamp_utc": "2026-04-13T06:33:54Z", "mode": "train", "global_step": 3294, "epoch": 0.33088900050226017, "loss": 0.041, "grad_norm": 6.109928131103516, "learning_rate": 2.1212121212121217e-08, "num_tokens": 6189614.0, "completions/mean_length": 124.0, "completions/min_length": 115.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 124.0, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.9894497990608215, "rewards/meter/std": 0.002960347104817629, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8836444020271301, "rewards/repeat_soft/std": 0.06809689849615097, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8096168637275696, "rewards/total_composite/std": 0.007265282329171896, "reward": 0.8096168637275696, "reward_std": 0.0072652786038815975, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07851393520832062, "sampling/sampling_logp_difference/max": 1.59437894821167, "sampling/importance_sampling_ratio/min": 0.20303459465503693, "sampling/importance_sampling_ratio/mean": 1.0130099058151245, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3382095657289028, "clip_ratio/low_mean": 0.036633451003581285, "clip_ratio/low_min": 0.036633451003581285, "clip_ratio/high_mean": 0.03176331939175725, "clip_ratio/high_max": 0.03176331939175725, "clip_ratio/region_mean": 0.06839677039533854, "reward_total_mean": 0.8096168637275696, "reward_meter_mean": 0.9894497990608215, "reward_meter_std": 0.002960347104817629, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8836444020271301, "reward_repeat_soft_std": 0.06809689849615097, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8096168637275696, "reward_total_composite_std": 0.007265282329171896} {"timestamp_utc": "2026-04-13T06:34:06Z", "mode": "train", "global_step": 3295, "epoch": 0.3309894525364139, "loss": -0.1452, "grad_norm": 1.5658140182495117, "learning_rate": 1.8181818181818185e-08, "num_tokens": 6191305.0, "completions/mean_length": 110.375, "completions/min_length": 46.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 53.000003814697266, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.8693521022796631, "rewards/meter/std": 0.34834134578704834, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9327556490898132, "rewards/repeat_soft/std": 0.04462124779820442, "rewards/judge_quality/mean": 0.3712499737739563, "rewards/judge_quality/std": 0.15037453174591064, "rewards/total_composite/mean": 0.7142065167427063, "rewards/total_composite/std": 0.28864678740501404, "reward": 0.7142065167427063, "reward_std": 0.28864678740501404, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09952045232057571, "sampling/sampling_logp_difference/max": 2.360100269317627, "sampling/importance_sampling_ratio/min": 0.09441076219081879, "sampling/importance_sampling_ratio/mean": 0.9975394010543823, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.33346888050436974, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0788243468850851, "clip_ratio/high_max": 0.0788243468850851, "clip_ratio/region_mean": 0.0788243468850851, "reward_total_mean": 0.7142065167427063, "reward_meter_mean": 0.8693521022796631, "reward_meter_std": 0.34834134578704834, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9327556490898132, "reward_repeat_soft_std": 0.04462124779820442, "reward_judge_quality_mean": 0.3712499737739563, "reward_judge_quality_std": 0.15037453174591064, "reward_total_composite_mean": 0.7142065167427063, "reward_total_composite_std": 0.28864678740501404} {"timestamp_utc": "2026-04-13T06:34:12Z", "mode": "train", "global_step": 3296, "epoch": 0.33108990457056753, "loss": -0.0098, "grad_norm": 15.083526611328125, "learning_rate": 1.5151515151515152e-08, "num_tokens": 6192593.0, "completions/mean_length": 22.0, "completions/min_length": 20.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.0, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.9815760850906372, "rewards/meter/std": 0.016209790483117104, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9546979069709778, "rewards/repeat_soft/std": 0.018489759415388107, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8131790161132812, "rewards/total_composite/std": 0.007160651963204145, "reward": 0.8131790161132812, "reward_std": 0.007160665467381477, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09378355741500854, "sampling/sampling_logp_difference/max": 1.313161849975586, "sampling/importance_sampling_ratio/min": 0.26896828413009644, "sampling/importance_sampling_ratio/mean": 0.9913434386253357, "sampling/importance_sampling_ratio/max": 1.7630507946014404, "entropy": 0.519037239253521, "clip_ratio/low_mean": 0.030059524346143007, "clip_ratio/low_min": 0.030059524346143007, "clip_ratio/high_mean": 0.06668313639238477, "clip_ratio/high_max": 0.06668313639238477, "clip_ratio/region_mean": 0.09674266073852777, "reward_total_mean": 0.8131790161132812, "reward_meter_mean": 0.9815760850906372, "reward_meter_std": 0.016209790483117104, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9546979069709778, "reward_repeat_soft_std": 0.018489759415388107, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8131790161132812, "reward_total_composite_std": 0.007160651963204145} {"timestamp_utc": "2026-04-13T06:34:18Z", "mode": "train", "global_step": 3297, "epoch": 0.33119035660472124, "loss": 0.0271, "grad_norm": 7.974599838256836, "learning_rate": 1.2121212121212122e-08, "num_tokens": 6194674.0, "completions/mean_length": 90.125, "completions/min_length": 86.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.125, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.9890191555023193, "rewards/meter/std": 0.003350069746375084, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9482499957084656, "rewards/repeat_soft/std": 0.057004060596227646, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.820758581161499, "rewards/total_composite/std": 0.03993517532944679, "reward": 0.820758581161499, "reward_std": 0.03993518650531769, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09183710068464279, "sampling/sampling_logp_difference/max": 2.043825626373291, "sampling/importance_sampling_ratio/min": 0.12953221797943115, "sampling/importance_sampling_ratio/mean": 0.9968867301940918, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43706731125712395, "clip_ratio/low_mean": 0.04324458772316575, "clip_ratio/low_min": 0.04324458772316575, "clip_ratio/high_mean": 0.03694559447467327, "clip_ratio/high_max": 0.03694559447467327, "clip_ratio/region_mean": 0.08019018219783902, "reward_total_mean": 0.820758581161499, "reward_meter_mean": 0.9890191555023193, "reward_meter_std": 0.003350069746375084, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9482499957084656, "reward_repeat_soft_std": 0.057004060596227646, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.820758581161499, "reward_total_composite_std": 0.03993517532944679} {"timestamp_utc": "2026-04-13T06:34:24Z", "mode": "train", "global_step": 3298, "epoch": 0.33129080863887495, "loss": 0.0839, "grad_norm": 13.742703437805176, "learning_rate": 9.090909090909092e-09, "num_tokens": 6196310.0, "completions/mean_length": 33.5, "completions/min_length": 27.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.5, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9588850736618042, "rewards/meter/std": 0.09258788079023361, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9545279741287231, "rewards/repeat_soft/std": 0.01623792015016079, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720350325107574, "rewards/total_composite/mean": 0.8427010774612427, "rewards/total_composite/std": 0.08652151376008987, "reward": 0.8427010774612427, "reward_std": 0.08652150630950928, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10500514507293701, "sampling/sampling_logp_difference/max": 0.9519586563110352, "sampling/importance_sampling_ratio/min": 0.4546465575695038, "sampling/importance_sampling_ratio/mean": 1.0190318822860718, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6753583662211895, "clip_ratio/low_mean": 0.08136310614645481, "clip_ratio/low_min": 0.08136310614645481, "clip_ratio/high_mean": 0.023863636888563633, "clip_ratio/high_max": 0.023863636888563633, "clip_ratio/region_mean": 0.10522674303501844, "reward_total_mean": 0.8427010774612427, "reward_meter_mean": 0.9588850736618042, "reward_meter_std": 0.09258788079023361, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9545279741287231, "reward_repeat_soft_std": 0.01623792015016079, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720350325107574, "reward_total_composite_mean": 0.8427010774612427, "reward_total_composite_std": 0.08652151376008987} {"timestamp_utc": "2026-04-13T06:34:35Z", "mode": "train", "global_step": 3299, "epoch": 0.33139126067302865, "loss": -0.1481, "grad_norm": 2.685288429260254, "learning_rate": 6.060606060606061e-09, "num_tokens": 6198504.0, "completions/mean_length": 160.25, "completions/min_length": 101.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 110.00000762939453, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.9715009331703186, "rewards/meter/std": 0.05505876988172531, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9701486825942993, "rewards/repeat_soft/std": 0.02906518615782261, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.281574010848999, "rewards/total_composite/mean": 0.8323152661323547, "rewards/total_composite/std": 0.10116632282733917, "reward": 0.8323152661323547, "reward_std": 0.10116631537675858, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0946456715464592, "sampling/sampling_logp_difference/max": 3.7493536472320557, "sampling/importance_sampling_ratio/min": 0.02353295125067234, "sampling/importance_sampling_ratio/mean": 0.9909297823905945, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.345858421176672, "clip_ratio/low_mean": 0.049230257980525494, "clip_ratio/low_min": 0.049230257980525494, "clip_ratio/high_mean": 0.028017994947731495, "clip_ratio/high_max": 0.028017994947731495, "clip_ratio/region_mean": 0.07724825292825699, "reward_total_mean": 0.8323152661323547, "reward_meter_mean": 0.9715009331703186, "reward_meter_std": 0.05505876988172531, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9701486825942993, "reward_repeat_soft_std": 0.02906518615782261, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.281574010848999, "reward_total_composite_mean": 0.8323152661323547, "reward_total_composite_std": 0.10116632282733917} {"timestamp_utc": "2026-04-13T06:34:44Z", "mode": "train", "global_step": 3300, "epoch": 0.3314917127071823, "loss": -0.0197, "grad_norm": 4.2629828453063965, "learning_rate": 3.0303030303030304e-09, "num_tokens": 6201825.0, "completions/mean_length": 207.125, "completions/min_length": 187.0, "completions/max_length": 216.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 207.125, "completions/min_terminated_length": 187.0, "completions/max_terminated_length": 216.0, "rewards/meter/mean": 0.9697142839431763, "rewards/meter/std": 0.047726232558488846, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6904106736183167, "rewards/repeat_soft/std": 0.06601713597774506, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.1249857097864151, "rewards/total_composite/mean": 0.7580375075340271, "rewards/total_composite/std": 0.04722398519515991, "reward": 0.7580375075340271, "reward_std": 0.04722397401928902, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05816280096769333, "sampling/sampling_logp_difference/max": 2.2263638973236084, "sampling/importance_sampling_ratio/min": 0.10792012512683868, "sampling/importance_sampling_ratio/mean": 0.9990186095237732, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.27459891326725483, "clip_ratio/low_mean": 0.016779666766524315, "clip_ratio/low_min": 0.016779666766524315, "clip_ratio/high_mean": 0.0408134707249701, "clip_ratio/high_max": 0.0408134707249701, "clip_ratio/region_mean": 0.05759313749149442, "reward_total_mean": 0.7580375075340271, "reward_meter_mean": 0.9697142839431763, "reward_meter_std": 0.047726232558488846, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6904106736183167, "reward_repeat_soft_std": 0.06601713597774506, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.1249857097864151, "reward_total_composite_mean": 0.7580375075340271, "reward_total_composite_std": 0.04722398519515991} {"timestamp_utc": "2026-04-13T06:35:36Z", "mode": "eval", "global_step": 3300, "epoch": 0.3314917127071823, "eval_loss": NaN, "eval_runtime": 51.682, "eval_samples_per_second": 1.548, "eval_steps_per_second": 0.193, "eval_num_tokens": 6201825.0, "eval_completions/mean_length": 105.825, "eval_completions/min_length": 41.9, "eval_completions/max_length": 229.0, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 95.55357208251954, "eval_completions/min_terminated_length": 41.9, "eval_completions/max_terminated_length": 165.7, "eval_rewards/meter/mean": 0.9251375734806061, "eval_rewards/meter/std": 0.14565520477481186, "eval_rewards/count_adherence/mean": 0.9975000023841858, "eval_rewards/count_adherence/std": 0.00707106739282608, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.07071067690849304, "eval_rewards/repeat_soft/mean": 0.9031356692314148, "eval_rewards/repeat_soft/std": 0.08098689429461955, "eval_rewards/judge_quality/mean": 0.4209999948740005, "eval_rewards/judge_quality/std": 0.11722671715542674, "eval_rewards/total_composite/mean": 0.7727779448032379, "eval_rewards/total_composite/std": 0.11504079140722752, "eval_reward": 0.7727779448032379, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.03713842239230871, "eval_sampling/sampling_logp_difference/max": 0.8526546716690063, "eval_sampling/importance_sampling_ratio/min": 0.4309925824403763, "eval_sampling/importance_sampling_ratio/mean": 1.0088137984275818, "eval_sampling/importance_sampling_ratio/max": 1.3825965166091918, "eval_entropy": 0.37584379613399505, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7727779448032379, "eval_reward_meter_mean": 0.9251375734806061, "eval_reward_meter_std": 0.14565520477481186, "eval_reward_count_adherence_mean": 0.9975000023841858, "eval_reward_count_adherence_std": 0.00707106739282608, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.07071067690849304, "eval_reward_repeat_soft_mean": 0.9031356692314148, "eval_reward_repeat_soft_std": 0.08098689429461955, "eval_reward_judge_quality_mean": 0.4209999948740005, "eval_reward_judge_quality_std": 0.11722671715542674, "eval_reward_total_composite_mean": 0.7727779448032379, "eval_reward_total_composite_std": 0.11504079140722752} {"timestamp_utc": "2026-04-13T06:35:39Z", "mode": "train", "global_step": 3300, "epoch": 0.3314917127071823, "train_runtime": 30072.346, "train_samples_per_second": 0.878, "train_steps_per_second": 0.11, "total_flos": 0.0, "train_loss": 0.006711615180229825}